From 581ff5fc73df3de6f3577f993735a3ea5692d757 Mon Sep 17 00:00:00 2001 From: diegosouzapw Date: Wed, 8 Apr 2026 13:40:04 -0300 Subject: [PATCH] docs: update system documentation and sync i18n for v3.5.5 --- AGENTS.md | 5 +- CHANGELOG.md | 22 + README.md | 20 +- docs/ARCHITECTURE.md | 8 +- docs/FEATURES.md | 28 +- docs/TROUBLESHOOTING.md | 54 + docs/i18n/ar/README.md | 2472 ++++++++-------- docs/i18n/ar/docs/ARCHITECTURE.md | 871 +++--- docs/i18n/ar/docs/FEATURES.md | 154 +- docs/i18n/ar/docs/TROUBLESHOOTING.md | 335 ++- docs/i18n/ar/llm.txt | 476 ++++ docs/i18n/bg/README.md | 2475 ++++++++-------- docs/i18n/bg/docs/ARCHITECTURE.md | 749 ++--- docs/i18n/bg/docs/FEATURES.md | 122 +- docs/i18n/bg/docs/TROUBLESHOOTING.md | 333 ++- docs/i18n/bg/llm.txt | 476 ++++ docs/i18n/cs/README.md | 2365 +++++++++------- docs/i18n/cs/docs/ARCHITECTURE.md | 737 ++--- docs/i18n/cs/docs/FEATURES.md | 122 +- docs/i18n/cs/docs/TROUBLESHOOTING.md | 300 +- docs/i18n/cs/llm.txt | 476 ++++ docs/i18n/da/README.md | 2357 +++++++++------- docs/i18n/da/docs/ARCHITECTURE.md | 717 ++--- docs/i18n/da/docs/FEATURES.md | 122 +- docs/i18n/da/docs/TROUBLESHOOTING.md | 298 +- docs/i18n/da/llm.txt | 476 ++++ docs/i18n/de/README.md | 2364 +++++++++------- docs/i18n/de/docs/ARCHITECTURE.md | 740 ++--- docs/i18n/de/docs/FEATURES.md | 122 +- docs/i18n/de/docs/TROUBLESHOOTING.md | 291 +- docs/i18n/de/llm.txt | 476 ++++ docs/i18n/es/README.md | 2417 +++++++++------- docs/i18n/es/docs/ARCHITECTURE.md | 675 +++-- docs/i18n/es/docs/FEATURES.md | 122 +- docs/i18n/es/docs/TROUBLESHOOTING.md | 254 +- docs/i18n/es/llm.txt | 476 ++++ docs/i18n/fi/README.md | 2369 +++++++++------- docs/i18n/fi/docs/ARCHITECTURE.md | 757 ++--- docs/i18n/fi/docs/FEATURES.md | 122 +- docs/i18n/fi/docs/TROUBLESHOOTING.md | 298 +- docs/i18n/fi/llm.txt | 476 ++++ docs/i18n/fr/README.md | 2334 ++++++++------- docs/i18n/fr/docs/ARCHITECTURE.md | 751 ++--- docs/i18n/fr/docs/FEATURES.md | 122 +- docs/i18n/fr/docs/TROUBLESHOOTING.md | 300 +- docs/i18n/fr/llm.txt | 476 ++++ docs/i18n/he/README.md | 21 +- docs/i18n/he/docs/ARCHITECTURE.md | 10 +- docs/i18n/he/docs/FEATURES.md | 30 +- docs/i18n/he/docs/TROUBLESHOOTING.md | 56 + docs/i18n/he/llm.txt | 476 ++++ docs/i18n/hi/README.md | 2506 ++++++++-------- docs/i18n/hi/docs/ARCHITECTURE.md | 747 ++--- docs/i18n/hi/docs/FEATURES.md | 122 +- docs/i18n/hi/docs/TROUBLESHOOTING.md | 297 +- docs/i18n/hi/llm.txt | 476 ++++ docs/i18n/hu/README.md | 2479 ++++++++-------- docs/i18n/hu/docs/ARCHITECTURE.md | 737 ++--- docs/i18n/hu/docs/FEATURES.md | 122 +- docs/i18n/hu/docs/TROUBLESHOOTING.md | 298 +- docs/i18n/hu/llm.txt | 476 ++++ docs/i18n/id/README.md | 2395 +++++++++------- docs/i18n/id/docs/ARCHITECTURE.md | 747 ++--- docs/i18n/id/docs/FEATURES.md | 122 +- docs/i18n/id/docs/TROUBLESHOOTING.md | 300 +- docs/i18n/id/llm.txt | 476 ++++ docs/i18n/in/README.md | 2510 +++++++++-------- docs/i18n/in/docs/ARCHITECTURE.md | 755 ++--- docs/i18n/in/docs/FEATURES.md | 122 +- docs/i18n/in/docs/TROUBLESHOOTING.md | 297 +- docs/i18n/in/llm.txt | 476 ++++ docs/i18n/it/README.md | 2377 +++++++++------- docs/i18n/it/docs/ARCHITECTURE.md | 747 ++--- docs/i18n/it/docs/FEATURES.md | 122 +- docs/i18n/it/docs/TROUBLESHOOTING.md | 300 +- docs/i18n/it/llm.txt | 476 ++++ docs/i18n/ja/README.md | 2480 ++++++++-------- docs/i18n/ja/docs/ARCHITECTURE.md | 751 ++--- docs/i18n/ja/docs/FEATURES.md | 122 +- docs/i18n/ja/docs/TROUBLESHOOTING.md | 289 +- docs/i18n/ja/llm.txt | 476 ++++ docs/i18n/ko/README.md | 2491 ++++++++-------- docs/i18n/ko/docs/ARCHITECTURE.md | 755 ++--- docs/i18n/ko/docs/FEATURES.md | 122 +- docs/i18n/ko/docs/TROUBLESHOOTING.md | 292 +- docs/i18n/ko/llm.txt | 476 ++++ docs/i18n/ms/README.md | 2357 +++++++++------- docs/i18n/ms/docs/ARCHITECTURE.md | 743 ++--- docs/i18n/ms/docs/FEATURES.md | 122 +- docs/i18n/ms/docs/TROUBLESHOOTING.md | 298 +- docs/i18n/ms/llm.txt | 476 ++++ docs/i18n/nl/README.md | 2327 ++++++++------- docs/i18n/nl/docs/ARCHITECTURE.md | 739 ++--- docs/i18n/nl/docs/FEATURES.md | 122 +- docs/i18n/nl/docs/TROUBLESHOOTING.md | 298 +- docs/i18n/nl/llm.txt | 476 ++++ docs/i18n/no/README.md | 2359 +++++++++------- docs/i18n/no/docs/ARCHITECTURE.md | 728 ++--- docs/i18n/no/docs/FEATURES.md | 122 +- docs/i18n/no/docs/TROUBLESHOOTING.md | 298 +- docs/i18n/no/llm.txt | 476 ++++ docs/i18n/phi/README.md | 2317 ++++++++------- docs/i18n/phi/docs/ARCHITECTURE.md | 675 +++-- docs/i18n/phi/docs/FEATURES.md | 118 +- docs/i18n/phi/docs/TROUBLESHOOTING.md | 298 +- docs/i18n/phi/llm.txt | 476 ++++ docs/i18n/pl/README.md | 2406 +++++++++------- docs/i18n/pl/docs/ARCHITECTURE.md | 753 ++--- docs/i18n/pl/docs/FEATURES.md | 122 +- docs/i18n/pl/docs/TROUBLESHOOTING.md | 300 +- docs/i18n/pl/llm.txt | 476 ++++ docs/i18n/pt-BR/README.md | 21 +- docs/i18n/pt-BR/docs/ARCHITECTURE.md | 10 +- docs/i18n/pt-BR/docs/FEATURES.md | 30 +- docs/i18n/pt-BR/docs/TROUBLESHOOTING.md | 56 + docs/i18n/pt-BR/llm.txt | 476 ++++ docs/i18n/pt/README.md | 2369 +++++++++------- docs/i18n/pt/docs/ARCHITECTURE.md | 751 ++--- docs/i18n/pt/docs/FEATURES.md | 122 +- docs/i18n/pt/docs/TROUBLESHOOTING.md | 300 +- docs/i18n/pt/llm.txt | 476 ++++ docs/i18n/ro/README.md | 2341 ++++++++------- docs/i18n/ro/docs/ARCHITECTURE.md | 731 ++--- docs/i18n/ro/docs/FEATURES.md | 122 +- docs/i18n/ro/docs/TROUBLESHOOTING.md | 298 +- docs/i18n/ro/llm.txt | 476 ++++ docs/i18n/ru/README.md | 2464 +++++++++------- docs/i18n/ru/docs/ARCHITECTURE.md | 749 ++--- docs/i18n/ru/docs/FEATURES.md | 122 +- docs/i18n/ru/docs/TROUBLESHOOTING.md | 300 +- docs/i18n/ru/llm.txt | 476 ++++ docs/i18n/sk/README.md | 2492 ++++++++-------- docs/i18n/sk/docs/ARCHITECTURE.md | 743 ++--- docs/i18n/sk/docs/FEATURES.md | 121 +- docs/i18n/sk/docs/TROUBLESHOOTING.md | 299 +- docs/i18n/sk/llm.txt | 476 ++++ docs/i18n/sv/README.md | 2349 ++++++++------- docs/i18n/sv/docs/ARCHITECTURE.md | 719 ++--- docs/i18n/sv/docs/FEATURES.md | 122 +- docs/i18n/sv/docs/TROUBLESHOOTING.md | 298 +- docs/i18n/sv/llm.txt | 476 ++++ docs/i18n/th/README.md | 2482 ++++++++-------- docs/i18n/th/docs/ARCHITECTURE.md | 749 ++--- docs/i18n/th/docs/FEATURES.md | 122 +- docs/i18n/th/docs/TROUBLESHOOTING.md | 298 +- docs/i18n/th/llm.txt | 476 ++++ docs/i18n/tr/README.md | 2399 +++++++++------- docs/i18n/tr/docs/ARCHITECTURE.md | 759 ++--- docs/i18n/tr/docs/FEATURES.md | 122 +- docs/i18n/tr/docs/TROUBLESHOOTING.md | 293 +- docs/i18n/tr/llm.txt | 476 ++++ docs/i18n/uk-UA/README.md | 2477 ++++++++-------- docs/i18n/uk-UA/docs/ARCHITECTURE.md | 750 ++--- docs/i18n/uk-UA/docs/FEATURES.md | 122 +- docs/i18n/uk-UA/docs/TROUBLESHOOTING.md | 298 +- docs/i18n/uk-UA/llm.txt | 476 ++++ docs/i18n/vi/README.md | 2479 ++++++++-------- docs/i18n/vi/docs/ARCHITECTURE.md | 745 ++--- docs/i18n/vi/docs/FEATURES.md | 122 +- docs/i18n/vi/docs/TROUBLESHOOTING.md | 300 +- docs/i18n/vi/llm.txt | 476 ++++ docs/i18n/zh-CN/README.md | 2471 ++++++++-------- docs/i18n/zh-CN/docs/ARCHITECTURE.md | 752 ++--- docs/i18n/zh-CN/docs/FEATURES.md | 122 +- docs/i18n/zh-CN/docs/TROUBLESHOOTING.md | 295 +- docs/i18n/zh-CN/llm.txt | 476 ++++ llm.txt | 26 +- open-sse/mcp-server/schemas/tools.ts | 2 + open-sse/mcp-server/tools/advancedTools.ts | 1 + open-sse/services/combo.ts | 47 +- open-sse/services/comboConfig.ts | 4 + open-sse/services/contextHandoff.ts | 383 +++ src/app/(dashboard)/dashboard/combos/page.tsx | 203 +- .../settings/components/ComboDefaultsTab.tsx | 83 +- src/app/api/settings/combo-defaults/route.ts | 3 + src/i18n/messages/en.json | 15 + src/i18n/messages/pt-BR.json | 15 + src/lib/db/contextHandoffs.ts | 167 ++ .../db/migrations/019_context_handoffs.sql | 28 + src/shared/constants/routingStrategies.ts | 8 + src/shared/schemas/validation.ts | 10 +- src/shared/validation/schemas.ts | 17 +- src/sse/handlers/chat.ts | 45 +- src/types/combo.ts | 2 +- src/types/settings.ts | 6 +- tests/unit/chat-context-relay.test.mjs | 197 ++ tests/unit/combo-config.test.mjs | 38 + tests/unit/combo-context-relay.test.mjs | 373 +++ tests/unit/context-handoff.test.mjs | 336 +++ 189 files changed, 79314 insertions(+), 45740 deletions(-) create mode 100644 docs/i18n/ar/llm.txt create mode 100644 docs/i18n/bg/llm.txt create mode 100644 docs/i18n/cs/llm.txt create mode 100644 docs/i18n/da/llm.txt create mode 100644 docs/i18n/de/llm.txt create mode 100644 docs/i18n/es/llm.txt create mode 100644 docs/i18n/fi/llm.txt create mode 100644 docs/i18n/fr/llm.txt create mode 100644 docs/i18n/he/llm.txt create mode 100644 docs/i18n/hi/llm.txt create mode 100644 docs/i18n/hu/llm.txt create mode 100644 docs/i18n/id/llm.txt create mode 100644 docs/i18n/in/llm.txt create mode 100644 docs/i18n/it/llm.txt create mode 100644 docs/i18n/ja/llm.txt create mode 100644 docs/i18n/ko/llm.txt create mode 100644 docs/i18n/ms/llm.txt create mode 100644 docs/i18n/nl/llm.txt create mode 100644 docs/i18n/no/llm.txt create mode 100644 docs/i18n/phi/llm.txt create mode 100644 docs/i18n/pl/llm.txt create mode 100644 docs/i18n/pt-BR/llm.txt create mode 100644 docs/i18n/pt/llm.txt create mode 100644 docs/i18n/ro/llm.txt create mode 100644 docs/i18n/ru/llm.txt create mode 100644 docs/i18n/sk/llm.txt create mode 100644 docs/i18n/sv/llm.txt create mode 100644 docs/i18n/th/llm.txt create mode 100644 docs/i18n/tr/llm.txt create mode 100644 docs/i18n/uk-UA/llm.txt create mode 100644 docs/i18n/vi/llm.txt create mode 100644 docs/i18n/zh-CN/llm.txt create mode 100644 open-sse/services/contextHandoff.ts create mode 100644 src/lib/db/contextHandoffs.ts create mode 100644 src/lib/db/migrations/019_context_handoffs.sql create mode 100644 tests/unit/chat-context-relay.test.mjs create mode 100644 tests/unit/combo-context-relay.test.mjs create mode 100644 tests/unit/context-handoff.test.mjs diff --git a/AGENTS.md b/AGENTS.md index 1e20037ee5..e2d957d5be 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -135,7 +135,8 @@ All persistence uses SQLite through domain-specific modules: `core.ts`, `providers.ts`, `models.ts`, `combos.ts`, `apiKeys.ts`, `settings.ts`, `backup.ts`, `proxies.ts`, `prompts.ts`, `webhooks.ts`, `detailedLogs.ts`, `domainState.ts`, `registeredKeys.ts`, `quotaSnapshots.ts`, `modelComboMappings.ts`, -`cliToolState.ts`, `encryption.ts`, `readCache.ts`, `secrets.ts`, `stateReset.ts`. +`cliToolState.ts`, `encryption.ts`, `readCache.ts`, `secrets.ts`, `stateReset.ts`, +`contextHandoffs.ts`. Schema migrations live in `db/migrations/` and run via `migrationRunner.ts`. `src/lib/localDb.ts` is a **re-export layer only** — never add logic there. @@ -189,7 +190,7 @@ Includes request/response translators with helpers for image handling. `autoCombo/`, `intentClassifier.ts`, `taskAwareRouter.ts`, `thinkingBudget.ts`, `contextManager.ts`, `modelDeprecation.ts`, `modelFamilyFallback.ts`, `emergencyFallback.ts`, `workflowFSM.ts`, `backgroundTaskDetector.ts`, `ipFilter.ts`, -`signatureCache.ts`, `volumeDetector.ts`, and more. +`signatureCache.ts`, `volumeDetector.ts`, `contextHandoff.ts`, and more. ### Domain Layer (`src/domain/`) diff --git a/CHANGELOG.md b/CHANGELOG.md index c0f5e46243..f852d7b2ab 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,28 @@ ## [Unreleased] +### ✨ New Features + +- **Context Relay Combo Strategy:** Added the new `context-relay` combo strategy with + priority-style routing, structured handoff summary generation once quota usage reaches + the warning threshold, and handoff injection after the next real account switch. +- **Global Context Relay Defaults:** Added global Settings defaults plus combo-level + configuration for `handoffThreshold`, `handoffModel`, and `handoffProviders`, so new or + unconfigured combos can inherit the feature consistently. + +### 🐛 Bug Fixes + +- **Context Relay In-Flight Deduplication:** Prevented duplicate handoff generation for + the same session/combo while an earlier summary request is still in flight. +- **Context Relay Provider Gating:** Aligned runtime behavior with configuration so + explicit `handoffProviders` exclusions, including an empty array, now disable handoff + generation as expected. + +### 📚 Documentation + +- **Context Relay Delivery Notes:** Documented the current architecture, runtime flow, and + Codex-focused scope in the feature docs, changelog, and agent guidance. + --- ## [3.5.4] — 2026-04-07 diff --git a/README.md b/README.md index b392f0cf73..1e09af1a31 100644 --- a/README.md +++ b/README.md @@ -243,7 +243,7 @@ Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Eve - **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention - **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI - **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next -- **Custom Combos** — Customizable fallback chains with 9 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random) +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) - **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard @@ -1304,7 +1304,17 @@ Then in `/dashboard/media` → **Transcription** tab: upload any audio or video ## 💡 Key Features -OmniRoute v2.0 is built as an operational platform, not just a relay proxy. +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. + +### 🆕 New — v3.5.5 Highlights (Apr 2026) + +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | ### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) @@ -1356,7 +1366,8 @@ OmniRoute v2.0 is built as an operational platform, not just a relay proxy. | 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | | 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | | 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | -| 🎨 **Custom Combos** | 9 balancing strategies + fallback chain control | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | | 🌐 **Wildcard Router** | `provider/*` dynamic routing | | 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | | 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | @@ -2183,9 +2194,10 @@ Se não quiser criar credenciais próprias agora, ainda é possível usar o flux | ---------------------------------------------- | --------------------------------------------------- | | [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | | [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | -| [MCP Server](open-sse/mcp-server/README.md) | 16 MCP tools, IDE configs, Python/TS/Go clients | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | | [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | | [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | | [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | | [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | | [Contributing](CONTRIBUTING.md) | Development setup and guidelines | diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 5db3edc0cc..c76f482a06 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -34,6 +34,7 @@ Core capabilities: - Anti-thundering herd protection with mutex locking - Signature-based request deduplication cache - Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity - Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) - Policy engine for centralized request evaluation (lockout → budget → fallback) - Request telemetry with p50/p95/p99 latency aggregation @@ -220,6 +221,8 @@ Services (business logic): - Wildcard model routing: `open-sse/services/wildcardRouter.ts` - Rate limit management: `open-sse/services/rateLimitManager.ts` - Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions Domain layer modules: @@ -800,7 +803,10 @@ Environment variables actively used by code: 5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. 6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). 7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). -8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. ## Operational Verification Checklist diff --git a/docs/FEATURES.md b/docs/FEATURES.md index dcea4eb0c5..a22f111002 100644 --- a/docs/FEATURES.md +++ b/docs/FEATURES.md @@ -16,7 +16,7 @@ Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI) ## 🎨 Combos -Create model routing combos with 6 strategies: priority, weighted, round-robin, random, least-used, and cost-optimized. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. ![Combos Dashboard](screenshots/02-combos.png) @@ -66,7 +66,7 @@ Comprehensive settings panel with tabs: - **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls - **Security** — API endpoint protection, custom provider blocking, IP filtering, session info - **Routing** — Model aliases, background task degradation -- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration - **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode ![Settings Dashboard](screenshots/06-settings.png) @@ -92,6 +92,30 @@ Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in a --- +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- + ## 🖼️ Media _(v2.0.3+)_ Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index d02d3e34dc..45f17164cd 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -15,6 +15,60 @@ Common problems and solutions for OmniRoute. | No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | | EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | | Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. --- diff --git a/docs/i18n/ar/README.md b/docs/i18n/ar/README.md index eca822e9ff..20338c5756 100644 --- a/docs/i18n/ar/README.md +++ b/docs/i18n/ar/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_وكيل واجهة برمجة التطبيقات العالمي الخاص بك — نقطة نهاية واحدة، وأكثر من 60 موفرًا، بدون أي توقف عن العمل. الآن مع**خادم MCP (25 أداة)**و**بروتوكول A2A**و**أنظمة الذاكرة/المهارات**و**تطبيق Electron Desktop**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**إكمالات الدردشة • التضمينات • إنشاء الصور • الفيديو • الموسيقى • الصوت • إعادة الترتيب •**بحث الويب**• خادم MCP • بروتوكول A2A • 100% TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _وكيل واجهة برمجة التطبيقات العالمي الخاص ب [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 موقع الويب](https://omniroute.online) • [🚀 البداية السريعة](#-بدء سريع) • [💡 الميزات](#-key-features) • [📖 المستندات](#-وثائق) • [💰 التسعير](#-تسعير في لمحة) • [💬 واتساب](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**متوفر باللغة:**🇺🇸 [الإنجليزية](README.md) | 🇧🇷 [البرتغالية (البرازيل)](docs/i18n/pt-BR/README.md) | 🇪🇸 [الإسبانية](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [الإيطالية](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [الألمانية](docs/i18n/de/README.md) | 🇮🇳 [خبر](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [بلغارسكي](docs/i18n/bg/README.md) | 🇩🇰 [الدانسك](docs/i18n/da/README.md) | 🇫🇮 [سومي](docs/i18n/fi/README.md) | 🇮🇱 [العربية](docs/i18n/he/README.md) | 🇭🇺 [المجرية](docs/i18n/hu/README.md) | 🇮🇩 [البهاسا الإندونيسية](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [البهاسا ملايو](docs/i18n/ms/README.md) | 🇳🇱 [هولندا](docs/i18n/nl/README.md) | 🇳🇴 [نورسك](docs/i18n/no/README.md) | 🇵🇹 [البرتغالية (البرتغال)](docs/i18n/pt/README.md) | 🇷🇴 [روماني](docs/i18n/ro/README.md) | 🇵🇱 [بولسكي](docs/i18n/pl/README.md) | 🇸🇰 [سلوفينسينا](docs/i18n/sk/README.md) | 🇸🇪 [السفينسكا](docs/i18n/sv/README.md) | 🇵🇭 [الفلبينية](docs/i18n/phi/README.md) | 🇨🇿 [تشيستينا](docs/i18n/cs/README.md)--- + + +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,554 +60,629 @@ _وكيل واجهة برمجة التطبيقات العالمي الخاص ب ## 📸 Dashboard Preview -<التفاصيل> +
+Click to see dashboard screenshots -انقر لرؤية لقطات شاشة لوحة التحكم +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | -| صفحة | لقطة شاشة | -| --------------------- | ------------------------------------------------- | ---------- | -| **مقدمو الخدمة** | ![Providers](docs/screenshots/01-providers.png) | -| **المجموعات** | ![Combos](docs/screenshots/02-combos.png) | -| **تحليلات** | ![تحليلات](docs/screenshots/03-analytics.png) | -| **الصحة** | ![الصحة](docs/screenshots/04-health.png) | -| **مترجم** | ![مترجم](docs/screenshots/05-translator.png) | -| **الإعدادات** | ![الإعدادات](docs/screenshots/06-settings.png) | -| **أدوات سطر الأوامر** | ![أدوات CLI](docs/screenshots/07-cli-tools.png) | -| **سجلات الاستخدام** | ![الاستخدام](docs/screenshots/08-usage.png) | -| **نقاط النهاية** | ![نقاط النهاية](docs/screenshots/09-endpoint.png) |
| + --- ### 🤖 Free AI Provider for your favorite coding agents -_قم بتوصيل أي أداة IDE أو CLI مدعومة بالذكاء الاصطناعي من خلال OmniRoute - بوابة واجهة برمجة التطبيقات المجانية للترميز غير المحدود._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ -<الجدول> -<تر> - - -OpenClaw
-أوبنكلاو -

-⭐ 205 ألف - - - -NanoBot
-نانوبوت -

-⭐ 20.9 ألف - - - -PicoClaw
-بيكوكلاو -

-⭐ 14.6 ألف - - - -ZeroClaw
-المخلب الصفري -

-⭐ 9.9 ألف - - - -IronClaw
-المخلب الحديدي -

-⭐ 2.1 كيلو - - -<تر> - - -OpenCode
-الرمز المفتوح -

-⭐ 106 كيلو - - - -Codex CLI
-Codex CLI -

-⭐ 60.8 ألف - - - -Claude Code
-كلود كود -

-⭐ 67.3 ألف - - - -Gemini CLI
-CLI الجوزاء -

-⭐ 94.7 ألف - - - -Kilo Code
-كود الكيلو -

-⭐ 15.5 ألف - - - + + + + + + + + + + + + + + + +
+ + OpenClaw
+ OpenClaw +

+ ⭐ 205K +
+ + NanoBot
+ NanoBot +

+ ⭐ 20.9K +
+ + PicoClaw
+ PicoClaw +

+ ⭐ 14.6K +
+ + ZeroClaw
+ ZeroClaw +

+ ⭐ 9.9K +
+ + IronClaw
+ IronClaw +

+ ⭐ 2.1K +
+ + OpenCode
+ OpenCode +

+ ⭐ 106K +
+ + Codex CLI
+ Codex CLI +

+ ⭐ 60.8K +
+ + Claude Code
+ Claude Code +

+ ⭐ 67.3K +
+ + Gemini CLI
+ Gemini CLI +

+ ⭐ 94.7K +
+ + Kilo Code
+ Kilo Code +

+ ⭐ 15.5K +
-📡 يتصل جميع الوكلاء عبر http://localhost:20128/v1 أو http://cloud.omniroute.online/v1 - تكوين واحد ونماذج وحصة غير محدودة--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**توقف عن إهدار المال وضرب الحدود:** +**Stop wasting money and hitting limits:** -- تنتهي صلاحية حصة الاشتراك غير المستخدمة كل شهر -- حدود المعدل تمنعك من الترميز المتوسط -- واجهات برمجة التطبيقات باهظة الثمن (20-50 دولارًا شهريًا لكل مزود) -- التبديل اليدوي بين مقدمي الخدمة +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute يحل هذا:** +**OmniRoute solves this:** -- ✅**تعظيم الاشتراكات**- تتبع الحصة، استخدم كل جزء منها قبل إعادة التعيين -- ✅**الرجوع التلقائي**- الاشتراك → مفتاح واجهة برمجة التطبيقات → رخيص → مجاني، بدون توقف -- ✅**حسابات متعددة**- جولة روبن بين الحسابات لكل مزود -- ✅**عالمي**- يعمل مع Claude Code وCodex وGemini CLI وCursor وCline وOpenClaw وأي أداة CLI--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**انضم إلى مجتمعنا!**[مجموعة WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — احصل على المساعدة وشارك النصائح وابق على اطلاع. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**الموقع الإلكتروني**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**المشاكل**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [مجموعة المجتمع](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**المساهمة**: راجع [CONTRIBUTING.md](CONTRIBUTING.md)، أو افتح علاقة عامة، أو اختر `العدد الأول الجيد` -**المشروع الأصلي**: [9router بواسطة decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -عند فتح مشكلة، يرجى تشغيل أمر معلومات النظام وإرفاق الملف الذي تم إنشاؤه:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -يؤدي هذا إلى إنشاء ملف "system-info.txt" مع إصدار Node.js، وإصدار OmniRoute، وتفاصيل نظام التشغيل، وأدوات CLI المثبتة (qoder، وgemini، و claude، وcodex، وantigravity، وdroid، وما إلى ذلك)، وحالة Docker/PM2، وحزم النظام - كل ما نحتاجه لإعادة إنتاج مشكلتك بسرعة. قم بإرفاق الملف مباشرة بمشكلة GitHub الخاصة بك.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**يواجه كل مطور يستخدم أدوات الذكاء الاصطناعي هذه المشكلات يوميًا.**تم تصميم OmniRoute لحلها جميعًا — بدءًا من تجاوز التكاليف وحتى الكتل الإقليمية، ومن تدفقات OAuth المعطلة إلى عمليات البروتوكول وإمكانية مراقبة المؤسسة. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. -<التفاصيل> -💸 1. "أدفع مقابل اشتراك باهظ الثمن ولكن لا يزال يتم مقاطعتي بسبب الحدود" +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -يدفع المطورون ما بين 20 إلى 200 دولار شهريًا مقابل Claude Pro أو Codex Pro أو GitHub Copilot. حتى عند الدفع، فإن الحصة لها حد أقصى — 5 ساعات من الاستخدام، أو حدود أسبوعية، أو حدود لسعر الدقيقة. في منتصف جلسة الترميز، يتوقف الموفر عن الاستجابة ويفقد المطور التدفق والإنتاجية. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**كيف يحل OmniRoute المشكلة:** +**How OmniRoute solves it:** --**الاحتياطي الذكي ذو 4 طبقات**— في حالة نفاد حصة الاشتراك، تتم إعادة التوجيه تلقائيًا إلى مفتاح واجهة برمجة التطبيقات ← رخيص ← مجاني بدون أي تدخل يدوي --**تتبع حدود الموفر**— يتم تحديث لقطات الحصص المخزنة مؤقتًا وفقًا لجدول من جانب الخادم (الافتراضي `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) مع توفر التحديث اليدوي في واجهة المستخدم --**دعم الحسابات المتعددة**— حسابات متعددة لكل مزود مع نظام روبن تلقائي — عند نفاد الحساب، يتم التبديل إلى التالي --**مجموعات مخصصة**— سلاسل احتياطية قابلة للتخصيص مع 9 إستراتيجيات موازنة (الأولوية، الموزونة، التعبئة أولاً، جولة روبن، P2C، عشوائي، الأقل استخدامًا، محسنة التكلفة، عشوائية صارمة) --**حصص الدستور الغذائي**— مراقبة حصص مساحة عمل الشركة/الفريق مباشرة في لوحة المعلومات
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard -<التفاصيل> -🔌 2. "أحتاج إلى استخدام عدة موفري خدمات ولكن لكل منهم واجهة برمجة تطبيقات مختلفة" + -يستخدم OpenAI تنسيقًا واحدًا، ويستخدم Claude (Anthropic) تنسيقًا آخر، ويستخدم Gemini تنسيقًا آخر. إذا أراد أحد المطورين اختبار النماذج من موفري خدمات مختلفين أو إجراء بديل فيما بينهم، فسيحتاج إلى إعادة تكوين مجموعات تطوير البرامج (SDK)، وتغيير نقاط النهاية، والتعامل مع التنسيقات غير المتوافقة. لدى موفري الخدمة المخصصين (FriendLI، NIM) نقاط نهاية نموذجية غير قياسية. +
+🔌 2. "I need to use multiple providers but each has a different API" -**كيف يحل OmniRoute المشكلة:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**نقطة النهاية الموحدة**— يعمل `http://localhost:20128/v1` كوكيل لجميع مقدمي الخدمة الذين يزيد عددهم عن 60 --**تنسيق الترجمة**— تلقائي وشفاف: OpenAI ↔ Claude ↔ Gemini ↔ Responses API --**تطهير الاستجابة**— إزالة الحقول غير القياسية (`x_groq`، `usage_breakdown`، `service_tier`) التي تكسر OpenAI SDK v1.83+ --**تطبيع الدور**— تحويل "المطور" → "النظام" لمقدمي الخدمات غير التابعين لـ OpenAI؛ "النظام" → "المستخدم" لـ GLM/ERNIE --**Think Tag Extraction**— يستخرج كتل `` من نماذج مثل DeepSeek R1 إلى ``reasoning_content'' القياسي --**الإخراج المنظم لـ Gemini**— التحويل التلقائي `json_schema` ← `responseMimeType`/`responseSchema` --**`stream` الافتراضي هو `false`**- يتماشى مع مواصفات OpenAI، ويتجنب SSE غير المتوقع في Python/Rust/Go SDKs
+**How OmniRoute solves it:** -<التفاصيل> -🌐 3. "يحظر مزود الذكاء الاصطناعي الخاص بي منطقتي/بلدي" +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -يقوم مقدمو الخدمة مثل OpenAI/Codex بحظر الوصول من مناطق جغرافية معينة. يحصل المستخدمون على أخطاء مثل `unsupported_country_region_territory` أثناء اتصالات OAuth وAPI. وهذا أمر محبط بشكل خاص للمطورين من البلدان النامية. + -**كيف يحل OmniRoute المشكلة:** +
+🌐 3. "My AI provider blocks my region/country" --**تكوين الوكيل ثلاثي المستوى**— وكيل قابل للتكوين على 3 مستويات: عالمي (كل حركة المرور)، لكل مزود (موفر واحد فقط)، ولكل اتصال/مفتاح --**شارات الوكيل المرمزة بالألوان**— المؤشرات المرئية: 🟢 الوكيل العالمي، 🟡 وكيل الموفر، 🔵 وكيل الاتصال، يظهر دائمًا عنوان IP --**تبادل رمز OAuth عبر الوكيل**— يمر تدفق OAuth أيضًا عبر الوكيل، مما يؤدي إلى حل مشكلة `unsupported_country_region_territory` --**اختبارات الاتصال عبر الوكيل**— تستخدم اختبارات الاتصال الوكيل الذي تم تكوينه (لا مزيد من التجاوز المباشر) --**دعم SOCKS5**— دعم وكيل SOCKS5 الكامل للتوجيه الخارجي --**انتحال بصمة إصبع TLS**— بصمة TLS تشبه المتصفح عبر `wreq-js` لتجاوز اكتشاف الروبوتات --**🔏 مطابقة بصمة CLI**— إعادة ترتيب الرؤوس وحقول النص لمطابقة التوقيعات الثنائية لـ CLI الأصلية، مما يقلل بشكل كبير من مخاطر الإبلاغ عن الحساب. يتم الحفاظ على عنوان IP الخاص بالوكيل — حيث يمكنك الحصول على إخفاء**و**IP في وقت واحد
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. -<التفاصيل> -🆓 4. "أريد استخدام الذكاء الاصطناعي في البرمجة ولكن ليس لدي المال" +**How OmniRoute solves it:** -لا يستطيع الجميع دفع ما بين 20 إلى 200 دولار شهريًا مقابل اشتراكات الذكاء الاصطناعي. يحتاج الطلاب والمطورون من البلدان الناشئة والهواة والمستقلون إلى الوصول إلى نماذج عالية الجودة بدون تكلفة. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**كيف يحل OmniRoute المشكلة:** + --**موفرو الطبقة المجانية المضمنون**— دعم أصلي لمقدمي الخدمة المجانية بنسبة 100%: Qoder (5 نماذج غير محدودة عبر OAuth: kimi-k2-thinking، qwen3-coder-plus، Deepseek-r1، minimax-m2، kimi-k2)، Qwen (4 نماذج غير محدودة: qwen3-coder-plus، qwen3-coder-flash، qwen3-coder-next، Vision-model)، Kiro (Claude + AWS Builder ID مجانًا)، Gemini CLI (180 ألف رمز مميز شهريًا مجانًا) --**Ollama Cloud**— نماذج Ollama المستضافة على السحابة على `api.ollama.com` مع فئة "الاستخدام الخفيف" مجانًا؛ استخدم البادئة `olmacloud/` --**المجموعات المجانية فقط**— السلسلة `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = 0 USD/الشهر بدون أي توقف عن العمل --**NVIDIA NIM Free Access**— ~40 دورة في الدقيقة وصول مجاني للأبد إلى أكثر من 70 نموذجًا على build.nvidia.com (الانتقال من الاعتمادات إلى حدود المعدل النقي) --**استراتيجية التكلفة المحسنة**— استراتيجية التوجيه التي تختار تلقائيًا أرخص مزود متاح +
+🆓 4. "I want to use AI for coding but I have no money" -<التفاصيل> -🔒 5. "أحتاج إلى حماية بوابة الذكاء الاصطناعي الخاصة بي من الوصول غير المصرح به" +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -عند تعريض بوابة AI للشبكة (LAN، VPS، Docker)، يمكن لأي شخص لديه العنوان استهلاك الرموز المميزة/الحصة النسبية للمطور. بدون الحماية، تكون واجهات برمجة التطبيقات (API) عرضة لإساءة الاستخدام والحقن الفوري وإساءة الاستخدام. +**How OmniRoute solves it:** -**كيف يحل OmniRoute المشكلة:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**إدارة مفاتيح واجهة برمجة التطبيقات**— الإنشاء والتدوير وتحديد النطاق لكل مزود من خلال صفحة `/dashboard/api-manager` المخصصة --**أذونات على مستوى النموذج**— تقييد مفاتيح واجهة برمجة التطبيقات (API) على نماذج محددة (`openai/*`، أنماط أحرف البدل)، مع تبديل السماح للكل/تقييد --**API Endpoint Protection**— اطلب مفتاحًا لـ `/v1/models` واحظر موفري خدمة محددين من القائمة --**Auth Guard + CSRF Protection**— جميع مسارات لوحة المعلومات محمية بالبرمجيات الوسيطة `withAuth` + رموز CSRF المميزة --**محدد المعدل**— تحديد معدل لكل IP مع نوافذ قابلة للتكوين --**تصفية IP**— القائمة المسموح بها/القائمة المحظورة للتحكم في الوصول --**حماية الحقن الفوري**— التعقيم ضد أنماط المطالبة الضارة --**تشفير AES-256-GCM**— بيانات الاعتماد مشفرة في حالة عدم النشاط
+ -<التفاصيل> -🛑 6. "تعطل مزود الخدمة الخاص بي وفقدت تدفق الترميز الخاص بي" +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -يمكن أن يصبح موفرو الذكاء الاصطناعي غير مستقرين، أو يعرضون أخطاء 5xx، أو يصلون إلى حدود المعدلات المؤقتة. إذا كان أحد المطورين يعتمد على موفر واحد، فسيتم مقاطعته. بدون قواطع الدائرة، يمكن أن تؤدي عمليات إعادة المحاولة المتكررة إلى تعطل التطبيق. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**كيف يحل OmniRoute المشكلة:** +**How OmniRoute solves it:** --**قاطع الدائرة لكل نموذج**— فتح/إغلاق تلقائي مع حدود قابلة للتكوين وفترة تهدئة (مغلق/مفتوح/نصف مفتوح)، محدد النطاق لكل نموذج لتجنب الكتل المتتالية --**التراجع الأسي**— تأخير إعادة المحاولة التدريجي --**مكافحة الرعد القطيع**— Mutex + حماية الإشارة ضد عواصف إعادة المحاولة المتزامنة --**السلاسل الاحتياطية المجمعة**— إذا فشل الموفر الأساسي، فسيتم دخوله تلقائيًا عبر السلسلة دون أي تدخل --**Combo Circuit Breaker**— التعطيل التلقائي لمقدمي الخدمات الفاشلين ضمن سلسلة التحرير والسرد --**لوحة معلومات الصحة**— مراقبة وقت التشغيل، وحالات قاطع الدائرة، وعمليات التأمين، وإحصائيات ذاكرة التخزين المؤقت، ووقت الاستجابة p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest -<التفاصيل> -🔧 7. "تكوين كل أداة من أدوات الذكاء الاصطناعي أمر ممل ومتكرر" + -يستخدم المطورون Cursor وClaude Code وCodex CLI وOpenClaw وGemini CLI وKilo Code... تحتاج كل أداة إلى تكوين مختلف (نقطة نهاية واجهة برمجة التطبيقات، المفتاح، النموذج). تعد إعادة التكوين عند تبديل مقدمي الخدمات أو النماذج مضيعة للوقت. +
+🛑 6. "My provider went down and I lost my coding flow" -**كيف يحل OmniRoute المشكلة:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**لوحة تحكم أدوات CLI**— صفحة مخصصة مع إعداد بنقرة واحدة لـ Claude Code، وCodex CLI، وOpenClaw، وKilo Code، وAntigravity، وCline --**GitHub Copilot Config Generator**— يُنشئ `chatLanguageModels.json` لرمز VS مع تحديد نموذج مجمع --**معالج الإعداد**— إعداد إرشادي من 4 خطوات للمستخدمين لأول مرة --**نقطة نهاية واحدة، جميع النماذج**— قم بتكوين `http://localhost:20128/v1` مرة واحدة، والوصول إلى أكثر من 60 موفرًا
+**How OmniRoute solves it:** -<التفاصيل> -🔑 8. "إدارة رموز OAuth المميزة من موفري خدمات متعددين أمر جحيم" +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code، وCodex، وGemini CLI، وCopilot — جميعهم يستخدمون OAuth 2.0 مع الرموز المميزة التي تنتهي صلاحيتها. يحتاج المطورون إلى إعادة المصادقة باستمرار، والتعامل مع "سر_العميل مفقود"، و"إعادة توجيه_uri_mismatch"، وحالات الفشل على الخوادم البعيدة. يمثل OAuth على LAN/VPS مشكلة بشكل خاص. + -**كيف يحل OmniRoute المشكلة:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**التحديث التلقائي للرمز المميز**— يتم تحديث رموز OAuth المميزة في الخلفية قبل انتهاء الصلاحية --**OAuth 2.0 (PKCE) مدمج**— التدفق التلقائي لـ Claude Code وCodex وGemini CLI وCopilot وKiro وQwen وQoder --**OAuth متعدد الحسابات**— حسابات متعددة لكل مزود عبر استخراج الرمز المميز JWT/ID --**OAuth LAN/Remote Fix**— اكتشاف IP الخاص لـ `redirect_uri` + وضع URL اليدوي للخوادم البعيدة --**OAuth Behind Nginx**— يستخدم window.location.origin للتوافق العكسي مع الوكيل --**دليل OAuth عن بعد**— دليل خطوة بخطوة لبيانات اعتماد Google Cloud على VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. -<التفاصيل> -📊 9. "لا أعرف كم أنفق أو أين" +**How OmniRoute solves it:** -يستخدم المطورون العديد من مقدمي الخدمات المدفوعة ولكن ليس لديهم رؤية موحدة للإنفاق. يمتلك كل مزود خدمة لوحة تحكم الفوترة الخاصة به، ولكن لا يوجد عرض موحد. التكاليف غير المتوقعة يمكن أن تتراكم. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**كيف يحل OmniRoute المشكلة:** + --**لوحة معلومات تحليلات التكلفة**— تتبع التكلفة لكل رمز مميز وإدارة الميزانية لكل مزود --**حدود الميزانية لكل طبقة**— سقف الإنفاق لكل طبقة يؤدي إلى حدوث تراجع تلقائي --**تكوين التسعير لكل نموذج**— أسعار قابلة للتكوين لكل نموذج --**إحصاءات الاستخدام لكل مفتاح API**— عدد الطلبات والطابع الزمني الأخير المستخدم لكل مفتاح --**لوحة التحكم التحليلية**— بطاقات الإحصائيات، ومخطط استخدام النموذج، وجدول الموفر مع معدلات النجاح وزمن الوصول +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" -<التفاصيل> -🐛 10. "لا أستطيع تشخيص الأخطاء والمشكلات في مكالمات الذكاء الاصطناعي" +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -عندما تفشل المكالمة، لا يعرف المطور ما إذا كان هناك حد للسعر، أو رمز مميز منتهي الصلاحية، أو تنسيق خاطئ، أو خطأ في الموفر. سجلات مجزأة عبر محطات مختلفة. وبدون إمكانية الملاحظة، يكون تصحيح الأخطاء عبارة عن تجربة وخطأ. +**How OmniRoute solves it:** -**كيف يحل OmniRoute المشكلة:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**لوحة تحكم السجلات الموحدة**— 4 علامات تبويب: سجلات الطلبات، وسجلات الوكيل، وسجلات التدقيق، ووحدة التحكم --**عارض سجل وحدة التحكم**— عارض بنمط المحطة الطرفية في الوقت الفعلي مع مستويات مرمزة بالألوان، والتمرير التلقائي، والبحث، والتصفية --**سجلات وكيل SQLite**— السجلات المستمرة التي تستمر حتى بعد إعادة تشغيل الخادم --**ساحة المترجم**— 4 أوضاع لتصحيح الأخطاء: ساحة اللعب (ترجمة التنسيق)، اختبار الدردشة (ذهابًا وإيابًا)، منصة الاختبار (دفعة)، المراقبة المباشرة (في الوقت الفعلي) --**قياس الطلب عن بعد**— زمن الاستجابة p50/p95/p99 + تتبع معرف طلب X --**التسجيل المستند إلى الملف مع التدوير**— يتم تدوير سجلات التطبيق حسب الحجم وأيام الاحتفاظ وعدد الأرشيف؛ يتم تدوير عناصر سجل المكالمات حسب أيام الاحتفاظ وعدد الملفات --**تقرير معلومات النظام**— يُنشئ `npm run system-info` ملف `system-info.txt` مع بيئتك الكاملة (إصدار Node، إصدار OmniRoute، نظام التشغيل، أدوات CLI، حالة Docker/PM2). قم بإرفاقه عند الإبلاغ عن مشكلات للفرز الفوري.
+ -<التفاصيل> -🏗️ 11. "إن نشر البوابة وصيانتها أمر معقد" +
+📊 9. "I don't know how much I'm spending or where" -يعد تثبيت وكيل AI وتكوينه وصيانته عبر بيئات مختلفة (محلية، VPS، Docker، سحابية) عملية كثيفة العمالة. مشاكل مثل المسارات المضمنة، EACCES في الدلائل، وتعارضات المنافذ، والبنيات عبر الأنظمة الأساسية تزيد من الاحتكاك. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**كيف يحل OmniRoute المشكلة:** +**How OmniRoute solves it:** --**تثبيت npm الشامل**— `npm install -g omniroute && omniroute` - تم --**منصة Docker المتعددة**— AMD64 + ARM64 الأصلي (Apple Silicon، AWS Graviton، Raspberry Pi) --**Docker Compose Profiles**— `base` (بدون أدوات CLI) و`cli` (مع Claude Code، وCodex، وOpenClaw) --**Electron Desktop App**— تطبيق أصلي لنظام التشغيل Windows/macOS/Linux مع علبة النظام، والتشغيل التلقائي، ووضع عدم الاتصال --**وضع المنفذ المقسم**— واجهة برمجة التطبيقات ولوحة المعلومات على منافذ منفصلة للسيناريوهات المتقدمة (الوكيل العكسي، وشبكات الحاويات) --**Cloud Sync**— مزامنة التكوين عبر الأجهزة عبر Cloudflare Workers --**النسخ الاحتياطية لقاعدة البيانات**— النسخ الاحتياطي التلقائي لجميع الإعدادات واستعادتها وتصديرها واستيرادها، باستخدام `DISABLE_SQLITE_AUTO_BACKUP` للنسخ الاحتياطية المُدارة خارجيًا
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency -<التفاصيل> -🌍 12. "الواجهة باللغة الإنجليزية فقط وفريقي لا يتحدث الإنجليزية" + -تواجه الفرق في البلدان غير الناطقة باللغة الإنجليزية، وخاصة في أمريكا اللاتينية وآسيا وأوروبا، صعوبة في التعامل مع الواجهات التي تستخدم اللغة الإنجليزية فقط. تعمل حواجز اللغة على تقليل الاعتماد وزيادة أخطاء التكوين. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**كيف يحل OmniRoute المشكلة:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**لوحة المعلومات i18n — 30 لغة**— أكثر من 500 مفتاح مترجم بما في ذلك العربية والبلغارية والدنماركية والألمانية والإسبانية والفنلندية والفرنسية والعبرية والهندية والمجرية والإندونيسية والإيطالية واليابانية والكورية والماليزية والهولندية والنرويجية والبولندية والبرتغالية (PT/BR) والرومانية والروسية والسلوفاكية والسويدية والتايلاندية والأوكرانية والفيتنامية والصينية والفلبينية والإنجليزية --**دعم RTL**— دعم من اليمين إلى اليسار للغتين العربية والعبرية --**الملفات التمهيدية متعددة اللغات**— 30 ترجمة كاملة للوثائق --**محدد اللغة**— رمز الكرة الأرضية في رأس الصفحة للتبديل في الوقت الفعلي
+**How OmniRoute solves it:** -<التفاصيل> -🔄 13. "أحتاج إلى أكثر من مجرد الدردشة - أحتاج إلى التضمين والصور والصوت" +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -الذكاء الاصطناعي ليس مجرد استكمال للدردشة. يحتاج المطورون إلى إنشاء صور، ونسخ الصوت، وإنشاء تضمينات لـ RAG، وإعادة ترتيب المستندات، والإشراف على المحتوى. تحتوي كل واجهة برمجة تطبيقات على نقطة نهاية وتنسيق مختلفين. + -**كيف يحل OmniRoute المشكلة:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Embeddings**— `/v1/embeddings` مع 6 موفري خدمات وأكثر من 9 نماذج --**إنشاء الصور**— `/v1/images/ Generations` مع 10 موفرين وأكثر من 20 نموذجًا (OpenAI، وxAI، وTogether، وFireworks، وNebius، وHyperbolic، وNanoBanana، وAntigravity، وSD WebUI، وComfyUI) --**تحويل النص إلى فيديو**— `/v1/videos/أجيال` — ComfyUI (AnimateDiff، SVD) وSD WebUI --**تحويل النص إلى موسيقى**— `/v1/music/generations` — ComfyUI (صوت ثابت مفتوح، MusicGen) --**نسخ الصوت**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM، HuggingFace، Qwen3 --**تحويل النص إلى كلام**— `/v1/audio/speech` — ElevenLabs، Nvidia NIM، HuggingFace، Coqui، Tortoise، Qwen3،**Inworld**،**Cartesia**،**PlayHT**، + مقدمي الخدمة الحاليين --**الإشراف**— `/v1/moderations` — التحقق من سلامة المحتوى --**إعادة الترتيب**— `/v1/rerank` — إعادة ترتيب مدى ملاءمة الوثيقة --**Responses API**— الدعم الكامل `/v1/responses` لـ Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. -<التفاصيل> -🧪 14. "ليس لدي طريقة لاختبار ومقارنة الجودة عبر النماذج" +**How OmniRoute solves it:** -يرغب المطورون في معرفة النموذج الأفضل لحالة الاستخدام الخاصة بهم - التعليمات البرمجية، والترجمة، والتفكير - ولكن المقارنة يدويًا بطيئة. لا توجد أدوات تقييم متكاملة. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**كيف يحل OmniRoute المشكلة:** + --**تقييمات LLM**— اختبار المجموعة الذهبية مع 10 حالات محملة مسبقًا تغطي التحيات، والرياضيات، والجغرافيا، وإنشاء التعليمات البرمجية، والامتثال لـ JSON، والترجمة، وتخفيض السعر، والرفض الآمن --**4 إستراتيجيات المطابقة**— `exact`، `contains`، `regex`، `custom` (وظيفة JS) --**منصة اختبار ساحة المترجم**— اختبار الدفعات بمدخلات متعددة ومخرجات متوقعة، ومقارنة بين الموفرين --**أداة اختبار الدردشة**— رحلة ذهابًا وإيابًا كاملة مع عرض الاستجابة المرئية --**المراقبة المباشرة**— البث المباشر لجميع الطلبات المتدفقة عبر الوكيل +
+🌍 12. "The interface is English-only and my team doesn't speak English" -<التفاصيل> -📈 15. "أحتاج إلى التوسع دون فقدان الأداء" +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -مع نمو حجم الطلب، يؤدي عدم التخزين المؤقت لنفس الأسئلة إلى توليد تكاليف مكررة. دون العجز، طلبات مكررة معالجة النفايات. يجب احترام حدود الأسعار لكل مزود. +**How OmniRoute solves it:** -**كيف يحل OmniRoute المشكلة:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**ذاكرة التخزين المؤقت الدلالية**— تعمل ذاكرة التخزين المؤقت ذات المستويين (التوقيع + الدلالي) على تقليل التكلفة ووقت الاستجابة --**صلاحية الطلب**— نافذة إلغاء البيانات المكررة لمدة 5 ثوانٍ للطلبات المتماثلة --**الكشف عن حدود المعدل**— عدد الدورات في الدقيقة لكل مزود، والفجوة الدنيا، والحد الأقصى للتتبع المتزامن --**حدود المعدل القابلة للتحرير**— الإعدادات الافتراضية القابلة للتكوين في الإعدادات → المرونة مع الثبات --**ذاكرة التخزين المؤقت للتحقق من صحة مفتاح واجهة برمجة التطبيقات**— ذاكرة تخزين مؤقت ثلاثية الطبقات لأداء الإنتاج --**لوحة معلومات الصحة مع القياس عن بعد**— زمن الاستجابة p50/p95/p99، وإحصائيات ذاكرة التخزين المؤقت، ووقت التشغيل
+ -<التفاصيل> -🤖 16. "أريد التحكم في سلوك النموذج عالميًا" +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -المطورون الذين يريدون جميع الاستجابات بلغة معينة، بنبرة معينة، أو يريدون الحد من الرموز المميزة للاستدلال. يعد تكوين هذا في كل أداة/طلب أمرًا غير عملي. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**كيف يحل OmniRoute المشكلة:** +**How OmniRoute solves it:** --**الحقن الفوري للنظام**— يتم تطبيق المطالبة العامة على جميع الطلبات --**التحقق من صحة ميزانية التفكير**— التحكم في تخصيص الرمز المميز لكل طلب (العبور، التلقائي، المخصص، التكيفي) --**9 استراتيجيات التوجيه**— استراتيجيات عالمية تحدد كيفية توزيع الطلبات --**Wildcard Router**— يتم توجيه أنماط `المزود/*` ديناميكيًا إلى أي مزود --**تبديل تمكين/تعطيل التحرير والسرد**— تبديل المجموعات مباشرة من لوحة المعلومات --**تبديل الموفر**— تمكين/تعطيل جميع اتصالات الموفر بنقرة واحدة --**موفري الخدمة المحظورون**— استبعاد موفري خدمة محددين من قائمة `/v1/models`
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex -<التفاصيل> -🧰 17. "أحتاج إلى أدوات MCP كقدرات منتج من الدرجة الأولى" + -تعرض العديد من بوابات الذكاء الاصطناعي MCP فقط كتفاصيل تنفيذ مخفية. تحتاج الفرق إلى طبقة تشغيل مرئية ويمكن التحكم فيها. +
+🧪 14. "I have no way to test and compare quality across models" -**كيف يحل OmniRoute المشكلة:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- يظهر MCP في لوحة التحكم وعلامة تبويب بروتوكول نقطة النهاية -- صفحة إدارة MCP مخصصة تحتوي على العمليات والأدوات والنطاقات والتدقيق -- بداية سريعة مدمجة لـ `omniroute --mcp` وتأهيل العميل
+**How OmniRoute solves it:** -<التفاصيل> -🧠 18. "أحتاج إلى تنسيق A2A مع مسارات المهام المتزامنة والدفقية" +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -تحتاج مسارات عمل الوكيل إلى ردود مباشرة وتنفيذ متدفق طويل الأمد مع التحكم في دورة الحياة. + -**كيف يحل OmniRoute المشكلة:** +
+📈 15. "I need to scale without losing performance" -- نقطة نهاية A2A JSON-RPC (`POST /a2a`) مع `message/send` و`message/stream` -- تدفق SSE مع انتشار الحالة الطرفية -- واجهات برمجة التطبيقات الخاصة بدورة حياة المهام لـ "المهام/الحصول" و"المهام/الإلغاء".
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. -<التفاصيل> -🛰️ 19. "أحتاج إلى صحة عملية MCP حقيقية، وليس حالة تخمينية" +**How OmniRoute solves it:** -تحتاج الفرق التشغيلية إلى معرفة ما إذا كان MCP حيًا بالفعل، وليس فقط ما إذا كان يمكن الوصول إلى واجهة برمجة التطبيقات (API). +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**كيف يحل OmniRoute المشكلة:** + -- ملف نبضات وقت التشغيل مع PID والطوابع الزمنية والنقل وعدد الأدوات ووضع النطاق -- واجهة برمجة تطبيقات حالة MCP التي تجمع بين نبضات القلب + النشاط الأخير -- بطاقات حالة واجهة المستخدم للعملية/وقت التشغيل/نضارة نبضات القلب +
+🤖 16. "I want to control model behavior globally" -<التفاصيل> -📋 20. "أحتاج إلى تنفيذ أداة MCP قابلة للتدقيق" +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -عندما تقوم الأدوات بتغيير التكوين أو تشغيل إجراءات العمليات، تحتاج الفرق إلى إمكانية التتبع الجنائي. +**How OmniRoute solves it:** -**كيف يحل OmniRoute المشكلة:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- تسجيل التدقيق المدعوم من SQLite لاستدعاءات أداة MCP -- عوامل التصفية حسب الأداة، والنجاح/الفشل، ومفتاح API، وترقيم الصفحات -- جدول تدقيق لوحة المعلومات + إحصائيات نقاط النهاية للأتمتة
+ -<التفاصيل> -🔐 21. "أحتاج إلى أذونات MCP محددة لكل عملية تكامل" +
+🧰 17. "I need MCP tools as first-class product capabilities" -يجب أن يتمتع العملاء المختلفون بإمكانية الوصول الأقل امتيازًا إلى فئات الأدوات. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**كيف يحل OmniRoute المشكلة:** +**How OmniRoute solves it:** -- 10 نطاقات MCP محببة للتحكم في الوصول إلى الأدوات -- إنفاذ النطاق والرؤية في واجهة مستخدم إدارة MCP -- الوضع الافتراضي الآمن للأدوات التشغيلية
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding -<التفاصيل> -⚙️ 22. "أحتاج إلى ضوابط تشغيلية دون إعادة الانتشار" + -تحتاج الفرق إلى تغييرات سريعة في وقت التشغيل أثناء الحوادث أو أحداث التكلفة. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**كيف يحل OmniRoute المشكلة:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- قم بتبديل تنشيط التحرير والسرد مباشرةً من لوحة معلومات MCP -- تطبيق ملفات تعريف المرونة من حزم السياسات المحددة مسبقًا -- إعادة ضبط حالة قاطع الدائرة من نفس لوحة العمليات
+**How OmniRoute solves it:** -<التفاصيل> -🔄 23. "أحتاج إلى رؤية وإلغاء مباشر لدورة حياة مهمة A2A" +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -وبدون رؤية دورة الحياة، يصبح من الصعب فرز حوادث المهام. + -**كيف يحل OmniRoute المشكلة:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- قائمة المهام/التصفية حسب الحالة/المهارة مع ترقيم الصفحات -- التعمق في البيانات الوصفية للمهمة، والأحداث، والتحف -- نقطة نهاية إلغاء المهمة وإجراء واجهة المستخدم مع التأكيد
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. -<التفاصيل> -🌊 24. "أحتاج إلى مقاييس تيار نشطة لتحميل A2A" +**How OmniRoute solves it:** -يتطلب تدفق سير العمل رؤية تشغيلية للتزامن والاتصالات المباشرة. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**كيف يحل OmniRoute المشكلة:** + -- عدادات التدفق النشطة مدمجة في حالة A2A -- الطابع الزمني للمهمة الأخيرة وعدد كل ولاية -- بطاقات لوحة القيادة A2A لمراقبة العمليات في الوقت الفعلي +
+📋 20. "I need auditable MCP tool execution" -<التفاصيل> -🪪 25. "أحتاج إلى اكتشاف وكيل قياسي للعملاء" +When tools mutate config or trigger ops actions, teams need forensic traceability. -يحتاج العملاء والمنسقون الخارجيون إلى بيانات تعريف يمكن قراءتها آليًا من أجل الإعداد. +**How OmniRoute solves it:** -**كيف يحل OmniRoute المشكلة:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- بطاقة الوكيل معروضة على `/.well-known/agent.json` -- القدرات والمهارات الموضحة في واجهة المستخدم الإدارية -- تتضمن واجهة برمجة التطبيقات لحالة A2A بيانات تعريف الاكتشاف للأتمتة
+ -<التفاصيل> -🧭 26. "أحتاج إلى إمكانية اكتشاف البروتوكول في تجربة المستخدم للمنتج" +
+🔐 21. "I need scoped MCP permissions per integration" -إذا لم يتمكن المستخدمون من اكتشاف أسطح البروتوكول، فسوف ينخفض جودة الاعتماد والدعم. +Different clients should have least-privilege access to tool categories. -**كيف يحل OmniRoute المشكلة:** +**How OmniRoute solves it:** -- صفحة**نقاط النهاية**الموحدة مع علامات تبويب Proxy وMCP وA2A وAPI Endpoints -- تبديل حالة الخدمة المضمنة (متصل/غير متصل) لـ MCP وA2A -- روابط من النظرة العامة إلى علامات تبويب الإدارة المخصصة
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling -<التفاصيل> -🧪 27. "أحتاج إلى التحقق من صحة البروتوكول الشامل مع عملاء حقيقيين" + -الاختبارات الوهمية ليست كافية للتحقق من توافق البروتوكول قبل الإصدار. +
+⚙️ 22. "I need operational controls without redeploying" -**كيف يحل OmniRoute المشكلة:** +Teams need quick runtime changes during incidents or cost events. -- مجموعة E2E التي تعمل على تشغيل التطبيق وتستخدم نقل عميل MCP SDK الحقيقي -- اختبارات عميل A2A لاكتشاف التدفقات وإرسالها ودفقها والحصول عليها وإلغائها -- التحقق من التأكيدات ضد تدقيق MCP وواجهات برمجة تطبيقات مهام A2A
+**How OmniRoute solves it:** -<التفاصيل> -📡 28. "أحتاج إلى إمكانية ملاحظة موحدة عبر جميع الواجهات" +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -يؤدي تقسيم إمكانية المراقبة حسب البروتوكول إلى إنشاء نقاط عمياء وMTTR أطول. + -**كيف يحل OmniRoute المشكلة:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- لوحات معلومات/سجلات/تحليلات موحدة في منتج واحد -- الصحة + التدقيق + طلب القياس عن بعد عبر طبقات OpenAI وMCP وA2A -- واجهات برمجة التطبيقات التشغيلية للحالة والأتمتة
+Without lifecycle visibility, task incidents become hard to triage. -<التفاصيل> -💼 29. "أحتاج إلى وقت تشغيل واحد للوكيل + الأدوات + تنسيق الوكيل" +**How OmniRoute solves it:** -يؤدي تشغيل العديد من الخدمات المنفصلة إلى زيادة تكلفة التشغيل وأوضاع الفشل. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**كيف يحل OmniRoute المشكلة:** + -- وكيل متوافق مع OpenAI وخادم MCP وخادم A2A في مكدس واحد -- المصادقة المشتركة والمرونة وتخزين البيانات وإمكانية الملاحظة -- نموذج سياسة متسق عبر جميع أسطح التفاعل +
+🌊 24. "I need active stream metrics for A2A load" -<التفاصيل> -🚀 30. "أحتاج إلى إرسال مهام سير عمل الوكيل دون امتداد التعليمات البرمجية اللاصقة" +Streaming workflows require operational insight into concurrency and live connections. -تفقد الفرق سرعتها عند دمج العديد من الخدمات والبرامج النصية المخصصة. +**How OmniRoute solves it:** -**كيف يحل OmniRoute المشكلة:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- استراتيجية نقطة النهاية الموحدة للعملاء والوكلاء -- واجهات مستخدم لإدارة البروتوكول مدمجة ومسارات التحقق من صحة الدخان -- أسس جاهزة للإنتاج (الأمان، التسجيل، المرونة، النسخ الاحتياطي)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**قواعد اللعبة أ: زيادة الاشتراك المدفوع إلى الحد الأقصى + نسخة احتياطية رخيصة**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -608,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**دليل التشغيل ب: مكدس البرمجة بدون تكلفة**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Playbook C: سلسلة احتياطية متاحة دائمًا على مدار 24 ساعة طوال أيام الأسبوع**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -631,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**قواعد اللعبة د: عمليات العميل مع MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> قم بإعداد ترميز الذكاء الاصطناعي في دقائق بسعر**$0/الشهر**. قم بتوصيل هذه الحسابات المجانية واستخدم المجموعة المدمجة**Free Stack**. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| خطوة | العمل | مقدمي الخدمات مقفلة | +| Step | Action | Providers Unlocked | | ---- | -------------------------------------------------- | ------------------------------------------------------------------ | -| 1 | الاتصال**Kiro**(معرف AWS Builder OAuth) | كلود سونيت 4.5، هايكو 4.5 —**غير محدود**| -| 2 | ربط**Qoder**(Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, Deepseek-r1... —**غير محدود**| -| 3 | ربط**كوين**(رمز الجهاز) | qwen3-coder-plus، qwen3-coder-flash... —**غير محدود**| -| 4 | الاتصال**Gemini CLI**(Google OAuth) | gemini-3-flash,gemini-2.5-pro —**180 ألف/الشهر مجانًا**| -| 5 | `/dashboard/combos` →**قالب مكدس مجاني ($0)**| جولة روبن لجميع مقدمي الخدمات المجانية تلقائيًا | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**قم بتوجيه أي IDE/CLI إلى:**`http://localhost:20128/v1` · مفتاح API: `any-string` · تم. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**تغطية إضافية اختيارية (مجانية أيضًا):**مفتاح Groq API (30 دورة في الدقيقة مجانًا)، NVIDIA NIM (40 دورة في الدقيقة مجانًا، أكثر من 70 طرازًا)، Cerebras (1 مليون tok/يوم)، مفتاح LongCat API (50 مليون رمز مميز/يوم!)، Cloudflare Workers AI (10 آلاف خلية عصبية/يوم، أكثر من 50 نموذجًا).## بداية سريعة +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## بداية سريعة ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **مستخدمي pnpm:**قم بتشغيل `pnpmوافق-builds -g` بعد التثبيت لتمكين البرامج النصية للبناء الأصلي المطلوبة من قبل `better-sqlite3` و`@swc/core`: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > -> ```باش -> تثبيت pnpm -g في كل الاتجاهات -> pnpm Approved-builds -g # حدد جميع الحزم → الموافقة -> الطريق الشامل +> ```bash +> pnpm install -g omniroute +> pnpm approve-builds -g # Select all packages → approve +> omniroute > ``` -تفتح لوحة المعلومات على `http://localhost:20128` ويكون عنوان URL الأساسي لواجهة برمجة التطبيقات هو `http://localhost:20128/v1`. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| الأمر | الوصف | -| ----------------------------- | ------------------------------------------------------------------------------------- | -| "الطريق الشامل" | بدء تشغيل الخادم (`PORT=20128` وواجهة برمجة التطبيقات ولوحة المعلومات على نفس المنفذ) | -| `الطريق الشامل --المنفذ 3000` | اضبط منفذ Canonical/API على 3000 | -| `الطريق الشامل --mcp` | بدء تشغيل خادم MCP (نقل stdio) | -| `الطريق الشامل --no-open` | لا تفتح المتصفح تلقائيًا | -| `الطريق الشامل --مساعدة` | عرض المساعدة | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -وضع المنفذ المقسم الاختياري:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -بالنسبة لمعظم عمليات النشر، تحتاج فقط إلى: +For most deployments, you only need: -| متغير | الافتراضي | الغرض | -| ------------------------ | ----------------------------- | -------------------------------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600000` | خط أساسي مشترك للجلب الأولي، ومهلات Undici المخفية، وطلبات بصمة TLS، ومهلة طلب/وكيل جسر واجهة برمجة التطبيقات | -| `STREAM_IDLE_TIMEOUT_MS` | يرث `REQUEST_TIMEOUT_MS` | الحد الأقصى للفجوة بين قطع الدفق قبل أن يقوم OmniRoute بإحباط دفق SSE | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -يتم الحفاظ على التوافق مع الإصدارات السابقة: لا تزال متغيرات مهلة `FETCH_TIMEOUT_MS` وAPI_BRIDGE_PROXY_TIMEOUT_MS الموجودة ومتغيرات المهلة الأخرى لكل طبقة تعمل وتتجاوز الخط الأساسي المشترك. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -تتوفر التجاوزات المتقدمة إذا كنت بحاجة إلى تحكم أفضل:| متغير | الافتراضي | الغرض | +Advanced overrides are available if you need finer control: + +| Variable | Default | Purpose | | ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | يرث `REQUEST_TIMEOUT_MS` | إجمالي مهلة طلب المنبع المستخدمة بواسطة إشارة إحباط الجلب الرئيسية | -| `FETCH_HEADERS_TIMEOUT_MS` | يرث `FETCH_TIMEOUT_MS` | الحد الزمني لـ Undici لتلقي رؤوس الاستجابة الأولية | -| `FETCH_BODY_TIMEOUT_MS` | يرث `FETCH_TIMEOUT_MS` | الحد الزمني Undici بين قطع النص الأساسي (`0` يعطله) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP مهلة الاتصال | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici مهلة مأخذ التوصيل الخامل | -| `TLS_CLIENT_TIMEOUT_MS` | يرث `FETCH_TIMEOUT_MS` | انتهت المهلة لطلبات بصمة TLS التي تم إجراؤها من خلال `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | يرث `REQUEST_TIMEOUT_MS` أو `30000` | انتهت المهلة لإعادة توجيه الوكيل `/v1` من منفذ API إلى منفذ لوحة المعلومات | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `الحد الأقصى (API_BRIDGE_PROXY_TIMEOUT_MS، 300000)` | انتهت مهلة الطلب الوارد على خادم جسر API | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | انتهت مهلة الرأس الوارد على خادم جسر API | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | مهلة البقاء على قيد الحياة على خادم جسر API | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | انتهت مهلة عدم نشاط مأخذ التوصيل على خادم جسر واجهة برمجة التطبيقات (`0` يعطله) | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -إذا قمت بتشغيل OmniRoute خلف Nginx أو Caddy أو Cloudflare أو وكيل عكسي آخر، فتأكد من الوكيل -تعد المهلات أيضًا أعلى من مهلات البث/الجلب في OmniRoute.### 2) Connect providers and create your API key +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. -1. افتح لوحة المعلومات → "الموفرون" وقم بتوصيل موفر واحد على الأقل (مفتاح OAuth أو API). -2. افتح لوحة المعلومات ← "نقاط النهاية" وأنشئ مفتاح واجهة برمجة التطبيقات. -3. (اختياري) افتح لوحة المعلومات → `المجموعات` وقم بتعيين السلسلة الاحتياطية.### 3) Point your coding tool to OmniRoute +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -يعمل مع Claude Code، وCodex CLI، وGemini CLI، وCursor، وCline، وOpenClaw، وOpenCode، وحزم SDK المتوافقة مع OpenAI.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (للعمليات التي تعتمد على الأدوات):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -ثم قم بتوصيل عميل MCP الخاص بك عبر أدوات "stdio" واختبار مثل: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (لسير العمل من وكيل إلى وكيل):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -760,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -يتحقق هذا الجناح من تدفقات عميل MCP وA2A الحقيقية مقابل تطبيق قيد التشغيل.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -768,14 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` -<التفاصيل> +
+Void Linux (`xbps-src` template) -إبطال Linux (قالب `xbps-src`) - -بالنسبة لمستخدمي Void Linux، يمكنك إنشاء حزمة أصلية باستخدام `xbps-src`. احفظ هذه الكتلة باسم `srcpkgs/omniroute/template`:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -787,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -795,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -871,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -882,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute متاح كصورة Docker عامة على [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**الجري السريع:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -892,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**مع ملف البيئة:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**استخدام Docker Compose:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -يتضمن دعم لوحة المعلومات لعمليات نشر Docker الآن نقرة واحدة**Cloudflare Quick Tunnel**على `Dashboard → Endpoints`. يقوم الأول بتمكين التنزيلات `cloudflared` فقط عند الحاجة، ويبدأ نفقًا مؤقتًا إلى نقطة النهاية `/v1` الحالية، ويعرض عنوان URL الذي تم إنشاؤه `https://*.trycloudflare.com/v1` مباشرةً أسفل عنوان URL العام العادي. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -ملاحظات: +Notes: -- عناوين URL للنفق السريع مؤقتة وتتغير بعد كل إعادة تشغيل. -- لا تتم استعادة الأنفاق السريعة تلقائيًا بعد إعادة تشغيل OmniRoute أو الحاوية. أعد تمكينها من لوحة التحكم عند الحاجة. -- التثبيت المُدار يدعم حاليًا Linux وmacOS وWindows على `x64` / `arm64`. -- الأنفاق السريعة المُدارة هي النقل الافتراضي عبر HTTP/2 لتجنب تحذيرات المخزن المؤقت QUIC UDP المزعجة في بيئات الحاويات المقيدة. قم بتعيين `CLOUDFLARED_PROTOCOL=quic` أو `auto` إذا كنت تريد وسيلة نقل مختلفة. -- تقوم صور Docker بتجميع جذور CA للنظام وتمريرها إلى `cloudflared` المُدارة، مما يتجنب فشل ثقة TLS عندما يبدأ النفق داخل الحاوية. -- يعمل SQLite في وضع WAL. يجب السماح لـ "docker stop" بالانتهاء حتى يتمكن OmniRoute من التحقق من أحدث التغييرات مرة أخرى في "storage.sqlite". -- قامت ملفات الإنشاء المجمعة بالفعل بتعيين فترة سماح للتوقف مدتها 40 ثانية. إذا قمت بتشغيل الصورة مباشرة، فاحتفظ بـ `--stop-timeout 40` (أو ما شابه) حتى لا تؤدي عمليات الإيقاف اليدوية إلى قطع عملية تنظيف إيقاف التشغيل. -- قم بتعيين `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` إذا كنت تريد أن يستخدم OmniRoute ملفًا ثنائيًا موجودًا بدلاً من تنزيله. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**استخدام Docker Compose مع Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -يمكن كشف OmniRoute بشكل آمن باستخدام توفير SSL التلقائي من Caddy. تأكد من أن سجل DNS A الخاص بنطاقك يشير إلى عنوان IP الخاص بخادمك.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` - -| صورة | العلامة | الحجم | الوصف | +| Image | Tag | Size | Description | | ------------------------ | -------- | ------ | --------------------- | -| `diegosouzapw/omniroute` | `الأحدث` | ~250 ميجابايت | أحدث إصدار مستقر | -| `diegosouzapw/omniroute` | `1.0.3` | ~250 ميجابايت | النسخة الحالية |--- +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | + +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**جديد!**OmniRoute متوفر الآن كتطبيق سطح مكتب أصلي**لأنظمة التشغيل Windows وmacOS وLinux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -قم بتشغيل OmniRoute كتطبيق مستقل لسطح المكتب - لا توجد محطة طرفية أو متصفح أو إنترنت مطلوب للطرز المحلية. يتضمن التطبيق المعتمد على Electron ما يلي: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**النافذة الأصلية**— نافذة تطبيق مخصصة مع تكامل علبة النظام -- 🔄**البدء التلقائي**— قم بتشغيل OmniRoute عند تسجيل الدخول إلى النظام -- 🔔**الإشعارات الأصلية**— احصل على تنبيهات بشأن استنفاد الحصص أو مشكلات المزود -- ⚡**التثبيت بنقرة واحدة**— NSIS (Windows)، DMG (macOS)، AppImage (Linux) -- 🌐**وضع عدم الاتصال بالإنترنت**— يعمل بشكل كامل دون اتصال بالإنترنت مع الخادم المُجمَّع### بداية سريعة +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### بداية سريعة ```bash # Development mode @@ -981,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -عند تصغيره، يظل OmniRoute موجودًا في علبة النظام لديك من خلال الإجراءات السريعة: +When minimized, OmniRoute lives in your system tray with quick actions: -- فتح لوحة القيادة -- تغيير منفذ الخادم -- قم بإنهاء التطبيق +- Open dashboard +- Change server port +- Quit application -📖 التوثيق الكامل: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| الطبقة | مقدم | التكلفة | إعادة ضبط الحصص | الأفضل لـ | -| ---------------------------------- | ---------------------------- | ------------------------------------- | ------------------------------------------- | ---------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳الإشتراك** | كلود كود (برو) | 20 دولارًا شهريًا | 5 ساعات + أسبوعي | اشتركت بالفعل | -| | الدستور الغذائي (زائد / برو) | 20-200 دولار شهريًا | 5 ساعات + أسبوعي | مستخدمي OpenAI | -| | الجوزاء CLI | **مجاني** | 180 ألف/شهر + 1 ألف/يوم | الجميع! | -| | جيثب مساعد الطيار | 10-19 دولارًا شهريًا | شهري | مستخدمي جيثب | -| **🔑 مفتاح واجهة برمجة التطبيقات** | نفيديا نيم | **مجانًا**(مطور للأبد) | ~40 دورة في الدقيقة | 70+ نماذج مفتوحة | -| | المخيخ | **مجانًا**(1 مليون توك/يوم) | 60 ألف دورة في الدقيقة / 30 دورة في الدقيقة | الأسرع في العالم | -| | جروك | **مجانًا**(30 دورة في الدقيقة) | 14.4K دورة في الدقيقة | لاما/جيما فائقة السرعة | -| | ديب سيك V3.2 | 0.27 دولار/1.10 دولار لكل مليون | لا شيء | أفضل منطق السعر/الجودة | -| | xAI Grok-4 سريع | **0.20 دولار/0.50 دولار لكل مليون**🆕 | لا شيء | أسرع + أداة استدعاء، منخفضة للغاية | -| | xAI Grok-4 (قياسي) | 0.20 دولار/1.50 دولار لكل مليون 🆕 | لا شيء | المنطق الرائد من xAI | -| | ميسترال | تجربة مجانية + مدفوعة | معدل محدود | الذكاء الاصطناعي الأوروبي | -| | اوبن راوتر | الدفع لكل استخدام | لا شيء | 100+ نماذج مجمعة. | -| **💰 رخيص** | GLM-5 (عبر Z.AI) 🆕 | 0.5 دولار/1 مليون | يوميا 10 صباحا | إخراج 128 كيلو، أحدث الرائد | -| | جي إل إم-4.7 | 0.6 دولار/1 مليون | يوميا 10 صباحا | نسخة احتياطية للميزانية | -| | ميني ماكس M2.5 🆕 | إدخال 0.3 دولار/1 مليون | المتداول لمدة 5 ساعات | الاستدلال + المهام الوكيلة | -| | ميني ماكس M2.1 | 0.2 دولار/1 مليون | المتداول لمدة 5 ساعات | الخيار الأرخص | -| | كيمي K2.5 (Moonshot API) 🆕 | الدفع لكل استخدام | لا شيء | الوصول المباشر إلى Moonshot API | -| | كيمي ك2 | 9 دولارات شهريًا مسطحة | 10 مليون رمز/شهر | التكلفة المتوقعة | -| **🆓مجانًا** | قدير | **$0** | غير محدود | 5 نماذج غير محدودة | -| | كوين | **$0** | غير محدود | 4 نماذج غير محدودة | -| | كيرو | **$0** | غير محدود | كلود سونيت/هايكو (AWS Builder) | -| | LongCat Flash-Lite 🆕 | **$0**(50 مليون توك/يوم 🔥) | 1 دورة في الثانية | أكبر حصة مجانية على وجه الأرض | -| | التلقيحات AI 🆕 | **$0**(لا حاجة لمفتاح) | 1 متطلب/15 ثانية | جي بي تي-5، كلود، ديب سيك، لاما 4 | -| | Cloudflare Workers AI 🆕 | **$0**(10 آلاف خلية عصبية/اليوم) | ~150 راحة/يوم | أكثر من 50 نموذجًا، حافة عالمية | -| | سكيليواي AI 🆕 | **$0**(إجمالي 1 مليون رمز) | معدل محدود | الاتحاد الأوروبي/اللائحة العامة لحماية البيانات، Qwen3 235B، Llama 70B | > 🆕**تمت إضافة نماذج جديدة (مارس 2026):**عائلة Grok-4 Fast بسعر 0.20 دولار أمريكي/0.50 دولار أمريكي/م (تم قياسها عند 1143 مللي ثانية - أسرع بنسبة 30% من Gemini 2.5 Flash)، GLM-5 عبر Z.AI بإخراج 128 ألف، واستدلال MiniMax M2.5، وتسعير DeepSeek V3.2 المحدث، وKimi K2.5 عبر Moonshot direct API. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 $0 Combo Stack — الإعداد المجاني الكامل:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**تكلفة صفر. لا تتوقف أبدًا عن البرمجة.**قم بتكوين هذا كمجموعة واحدة من OmniRoute وستحدث جميع الإجراءات الاحتياطية تلقائيًا - لا يوجد تبديل يدوي على الإطلاق.--- +--- --- ## 🆓 Free Models — What You Actually Get -> جميع الموديلات أدناه**مجانية بنسبة 100% ولا تتطلب أي بطاقة ائتمان**. يقوم OmniRoute بالمسارات التلقائية بينهما عند نفاد حصة واحدة - اجمعها جميعًا للحصول على مجموعة غير قابلة للكسر بقيمة 0 دولار.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| نموذج | البادئة | الحد | حد السعر | +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | --------------------- | -| `كلود-السوناتة-4.5` | `كر/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى اليومي | -| `كلود-هايكو-4.5` | `كر/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى اليومي | -| `كلود-أوبوس-4.6` | `كر/` |**غير محدود**| أحدث أعمال أوبوس عبر كيرو |### 🟢 QODER MODELS (Free PAT via qodercli) +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -| نموذج | البادئة | الحد | حد السعر | +### 🟢 QODER MODELS (Free PAT via qodercli) + +| Model | Prefix | Limit | Rate Limit | | ------------------ | ------ | ------------- | --------------- | -| `تفكير كيمي-ك2` | `إذا/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى | -| `qwen3-coder-plus` | `إذا/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى | -| `ديبسيك-R1` | `إذا/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى | -| `مينيماكس-m2.1` | `إذا/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى | -| `كيمي-k2` | `إذا/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -> طريقة الاتصال الموصى بها:**رمز الوصول الشخصي + `qodercli`**. متصفح OAuth هو -> تجريبي ومعطل افتراضيًا ما لم يتم تكوين متغيرات البيئة `QODER_OAUTH_*`.### 🟡 QWEN MODELS (Device Code Auth) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| نموذج | البادئة | الحد | حد السعر | +### 🟡 QWEN MODELS (Device Code Auth) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | ------------------- | -| `qwen3-coder-plus` | `س/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى | -| `qwen3-coder-flash` | `س/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى | -| `qwen3-coder-next` | `س/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى | -| `نموذج الرؤية` | `س/` |**غير محدود**| الوسائط المتعددة (صور) |### 🟣 GEMINI CLI (Google OAuth) +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| نموذج | البادئة | الحد | حد السعر | +### 🟣 GEMINI CLI (Google OAuth) + +| Model | Prefix | Limit | Rate Limit | | ------------------------ | ------ | --------------------------- | ------------- | -| `الجوزاء-3-معاينة فلاش` | `جي سي/` |**180 ألف توك/شهر**+ 1 ألف/يوم | إعادة الضبط الشهرية | -| `الجوزاء-2.5-برو` | `جي سي/` | 180 ألف/شهر (مسبح مشترك) | جودة عالية |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -| الطبقة | الحد اليومي | حد السعر | ملاحظات | +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---------- | ------------ | ----------- | ------------------------------------------------------ | -| مجاني (ديف) | لا يوجد غطاء رمزي |**~40 دورة في الدقيقة**| أكثر من 70 نموذجًا؛ الانتقال إلى حدود المعدل النقي منتصف عام 2025 | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -النماذج المجانية المشهورة: `moonshotai/kimi-k2.5` (Kimi K2.5)، `z-ai/glm4.7` (GLM 4.7)، `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2)، `nvidia/llama-3.3-70b-instruct`، `deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -| الطبقة | الحد اليومي | حد السعر | ملاحظات | +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) + +| Tier | Daily Limit | Rate Limit | Notes | | ---- | ----------------- | ---------------- | ------------------------------------------- | -| مجاني |**1 مليون قطعة/يوم**| 60 ألف دورة في الدقيقة / 30 دورة في الدقيقة | أسرع استنتاج LLM في العالم؛ يعيد يوميا | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -متاح مجانًا: `llama-3.3-70b`، `llama-3.1-8b`، `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -| الطبقة | الحد اليومي | حد السعر | ملاحظات | +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---- | ------------- | ---------------- | ----------------------------------------- | -| مجاني |**14.4 كيلو دورة في الدقيقة**| 30 دورة في الدقيقة لكل موديل | لا توجد بطاقة ائتمان؛ 429 على الحد، غير مشحونة | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -متاحة مجانًا: `llama-3.3-70b-versatile`، `gemma2-9b-it`، `mixtral-8x7b`، `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -| نموذج | البادئة | الحصة اليومية المجانية | ملاحظات | +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 + +| Model | Prefix | Daily Free Quota | Notes | | ----------------------------- | ------ | ----------------- | ----------------------- | -| `لونجكات-فلاش-لايت` | `لك/` |**50 مليون رمز**💥 | أكبر حصة مجانية على الإطلاق | -| `LongCat-Flash-Chat` | `لك/` | 500 ألف رمز | دردشة متعددة المنعطفات | -| ``التفكير الخاطف الطويل`` | `لك/` | 500 ألف رمز | الاستدلال / CoT | -| `لونجكات-فلاش-التفكير-2601` | `لك/` | 500 ألف رمز | نسخة يناير 2026 | -| `لونج كات-فلاش-أومني-2603` | `لك/` | 500 ألف رمز | الوسائط المتعددة | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | -> مجاني 100% أثناء وجودك في النسخة التجريبية العامة. قم بالتسجيل في [longcat.chat](https://longcat.chat) باستخدام البريد الإلكتروني أو الهاتف. تتم إعادة الضبط يوميًا في تمام الساعة 00:00 بالتوقيت العالمي المنسق.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. -| نموذج | البادئة | حد السعر | مقدم خلف | +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| `أوبيني` | `بول/` | 1 متطلب/15 ثانية | جي بي تي-5 | -| "كلود" | `بول/` | 1 متطلب/15 ثانية | أنثروبي كلود | -| `الجوزاء` | `بول/` | 1 متطلب/15 ثانية | جوجل الجوزاء | -| `البحث العميق` | `بول/` | 1 متطلب/15 ثانية | ديب سيك V3 | -| اللاما | `بول/` | 1 متطلب/15 ثانية | ميتا لاما 4 كشاف | -| `ميسترال` | `بول/` | 1 متطلب/15 ثانية | ميسترال لمنظمة العفو الدولية | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**بدون احتكاك:**لا يوجد اشتراك، ولا يوجد مفتاح API. أضف موفر التلقيح بحقل مفتاح فارغ وسيعمل على الفور.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| الطبقة | الخلايا العصبية اليومية | الاستخدام المعادل | ملاحظات | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 + +| Tier | Daily Neurons | Equivalent Usage | Notes | | ---- | ------------- | --------------------------------------- | ----------------------- | -| مجاني |**10,000**| ~150 LLM resp / 500 ثانية صوت / 15 ألف تضمين | الحافة العالمية، أكثر من 50 نموذجًا | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -النماذج المجانية الشهيرة: `@cf/meta/llama-3.3-70b-instruct`، `@cf/google/gemma-3-12b-it`، `@cf/openai/whisper-large-v3-turbo` (صوت مجاني!)، `@cf/qwen/qwen2.5-coder-15b-instruct` +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -> يتطلب رمز API المميز + معرف الحساب من [dash.cloudflare.com](https://dash.cloudflare.com). قم بتخزين معرف الحساب في إعدادات الموفر.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -| الطبقة | حصة مجانية | الموقع | ملاحظات | +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | | ---- | ------------- | ------------ | ----------------------------------- | -| مجاني |**مليون قطعة**| 🇫🇷 باريس، الاتحاد الأوروبي | لا حاجة لبطاقة الائتمان ضمن الحدود | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | -متاح مجانًا: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!) +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` -> متوافقة مع الاتحاد الأوروبي/اللائحة العامة لحماية البيانات. احصل على مفتاح واجهة برمجة التطبيقات على [console.scaleway.com](https://console.scaleway.com). +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). ->**💡 المجموعة المجانية المطلقة (11 مقدمًا، 0 دولار للأبد):** +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> كيرو (kr/) → كلود سونيت/هايكو غير محدود -> Qoder (if/) → kimi-k2-thinking، qwen3-coder-plus، Deepseek-r1 غير محدود -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 مليون رمز/يوم 🔥 -> التلقيح (pol/) → GPT-5، Claude، DeepSeek، Llama 4 - لا حاجة إلى مفتاح -> Qwen (qw/) → نماذج qwen3-coder غير محدودة -> Gemini (gemini/) → Gemini 2.5 Flash — 1500 طلب/يوم مجانًا -> Cloudflare AI (cf/) → أكثر من 50 نموذجًا - 10 آلاف خلية عصبية/اليوم -> Scaleway (scw/) → Qwen3 235B، Llama 70B — مليون رمز مجاني (الاتحاد الأوروبي) -> Groq (groq/) → Llama/Gemma — 14.4 ألف طلب/يوم بسرعة فائقة -> NVIDIA NIM (nvidia/) → أكثر من 70 طرازًا مفتوحًا - 40 دورة في الدقيقة إلى الأبد -> المخيخ (cerebras/) → اللاما/كوين الأسرع في العالم — مليون توك/اليوم -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> قم بنسخ أي صوت/فيديو مقابل**$0**— تقدم Deepgram مبلغًا مجانيًا بقيمة 200 دولار أمريكي، ونسخة احتياطية من AssemblyAI بقيمة 50 دولارًا أمريكيًا، وGroq Whisper كنسخة احتياطية غير محدودة للطوارئ. +## 🎙️ Free Transcription Combo -| مقدم | اعتمادات مجانية | أفضل موديل | حد السعر | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. + +| Provider | Free Credits | Best Model | Rate Limit | | ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | -| 🟢**ديبجرام**|**200 دولار مجانًا**(اشتراك) | `nova-3` — أفضل دقة، أكثر من 30 لغة | لا يوجد حد لعدد RPM على الاعتمادات المجانية | -| 🔵**AssemblyAI**|**50 دولارًا مجانًا**(اشتراك) | `universal-3-pro` - الفصول، المشاعر، معلومات تحديد الهوية الشخصية | لا يوجد حد لعدد RPM على الاعتمادات المجانية | -| 🔴**جروق**|**مجاني للأبد**| `whisper-large-v3` — OpenAI Whisper | 30 دورة في الدقيقة (معدل محدود) | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | -**التحرير والسرد المقترح في `/dashboard/combos`:**``` +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -ثم في `/dashboard/media` → علامة التبويب**Transcription**: قم بتحميل أي ملف صوت أو فيديو ← حدد نقطة نهاية التحرير والسرد الخاصة بك ← احصل على النسخ بتنسيقات مدعومة.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -تم تصميم OmniRoute v2.0 كمنصة تشغيلية، وليس مجرد وكيل ترحيل.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| ميزة | ماذا يفعل | -| ------------------------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Grok-4 Fast Family** | طرازات xAI بسعر 0.20 دولارًا أمريكيًا/0.50 دولارًا أمريكيًا للمتر المربع - تم قياسها بـ 1143 مللي ثانية (أسرع بنسبة 30% من Gemini 2.5 Flash) | -| 🧠**GLM-5 عبر Z.AI** | سياق إخراج 128 ألفًا، 0.5 دولار أمريكي/1 مليون — أحدث منتج رئيسي من عائلة GLM | -| 🔮**ميني ماكس M2.5** | الاستدلال + المهام الوكيلة بسعر 0.30 دولارًا أمريكيًا/مليون واحد — ترقية كبيرة من M2.1 | -| 🎯**أداة استدعاء العلم لكل نموذج** | لكل نموذج `toolCalling: true/false` في التسجيل - يتخطى AutoCombo النماذج التي لا تحتوي على أدوات | -| 🌍**كشف النوايا المتعددة اللغات** | الكلمات الأساسية PT/ZH/ES/AR في تسجيل AutoCombo — اختيار نموذج أفضل للمحتوى غير الإنجليزي | -| 📊**الإجراءات الاحتياطية المستندة إلى المعايير** | زمن استجابة حقيقي p95 من الطلبات المباشرة يغذي تسجيل التحرير والسرد - يتعلم AutoCombo من البيانات الفعلية | -| 🔁**طلب إلغاء البيانات المكررة** | نافذة إلغاء البيانات المستندة إلى تجزئة المحتوى — آمنة متعددة الوكلاء، وتمنع الرسوم المكررة | -| 🔌**استراتيجية جهاز التوجيه القابل للتوصيل** | واجهة "RouterStrategy" القابلة للتوسيع - أضف منطق توجيه مخصص كمكونات إضافية | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| ميزة | ماذا يفعل | -| -------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**ساحة اللعب النموذجية** | صفحة لوحة التحكم لاختبار أي نموذج مباشرة - محددات الموفر/النموذج/نقطة النهاية، محرر موناكو، البث، الإجهاض، التوقيت | -| 🔏**مطابقة بصمة CLI** | ترتيب الرأس/النص لكل موفر لمطابقة توقيعات CLI الأصلية - قم بالتبديل لكل موفر في الإعدادات > الأمان.**يتم الاحتفاظ بـ IP الوكيل الخاص بك** | -| 🤝**دعم ACP (بروتوكول العميل الوكيل)** | اكتشاف وكيل CLI (Codex، Claude، Goose، Gemini CLI، OpenClaw + 9 آخرين)، مولد العمليات، `/api/acp/agents` نقطة النهاية | -| 🤖**لوحة تحكم وكلاء ACP** | التصحيح › صفحة الوكلاء - شبكة مكونة من 14 وكيلًا مع حالة التثبيت والإصدار ونموذج الوكيل المخصص لأي أداة CLI. يحصل مستخدمو**OpenCode**على زر "تنزيل opencode.json" الذي يقوم تلقائيًا بإنشاء تكوين جاهز للاستخدام مع جميع الطرز المتاحة. | -| 🔧**توجيه نموذج مخصص `apiFormat`** | النماذج المخصصة ذات `apiFormat: "responses"` توجه الآن بشكل صحيح إلى مترجم Responses API | -| 🏢**عزل مساحة عمل الدستور الغذائي** | مساحات عمل Codex متعددة لكل بريد إلكتروني - يفصل OAuth الاتصالات بشكل صحيح عن طريق معرف مساحة العمل | -| 🔄**التحديث التلقائي الإلكتروني** | يتحقق تطبيق سطح المكتب من التحديثات + التثبيت التلقائي عند إعادة التشغيل | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| ميزة | ماذا يفعل | -| -------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------- | -| 🔧**خادم MCP (25 أداة)** | أدوات IDE/agent عبر 3 وسائل نقل: stdio، وSSE (`/api/mcp/sse`)، وHTTP القابل للتدفق (`/api/mcp/stream`). 18 نواة + 3 ذاكرة + 4 أدوات مهارات | -| 🤝**خادم A2A (JSON-RPC + SSE)** | تنفيذ المهام من وكيل إلى وكيل مع تدفقات المزامنة والتدفق | -| 🧭**صفحة نقاط النهاية الموحدة** | صفحة إدارة مبوبة مع علامات تبويب Endpoint Proxy وMCP وA2A وAPI Endpoints | -| 🎚️**تبديل تمكين / تعطيل الخدمة** | مفاتيح التشغيل/الإيقاف لـ MCP وA2A مع ثبات الإعدادات (الافتراضي: OFF) | -| 🛰️**نبضات وقت تشغيل MCP** | حالة العملية الحقيقية (معرف المنتج، وقت التشغيل، عمر نبضات القلب، النقل، وضع النطاق) | -| 📋**مسار تدقيق MCP** | سجلات التدقيق القابلة للتصفية مع النجاح/الفشل والإسناد الرئيسي | -| 🔐**تنفيذ نطاق MCP** | 10 أذونات نطاق تفصيلية للوصول إلى الأدوات الخاضعة للرقابة | -| 📡**إدارة دورة حياة المهام A2A** | قائمة/تصفية المهام، فحص الأحداث/التحف، إلغاء المهام قيد التشغيل | -| 📋**اكتشاف بطاقة الوكيل** | `/.well-known/agent.json` للاكتشاف التلقائي للعميل | -| 🧪**أداة اختبار البروتوكول E2E** | يتدفق عميل MCP SDK + A2A الحقيقي في "اختبار: البروتوكولات: e2e" | -| ⚙️**ضوابط التشغيل** | مجموعة التبديل، وتطبيق ملفات تعريف المرونة، وإعادة ضبط القواطع من سطح تحكم واحد | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| ميزة | ماذا يفعل | -| ---------------------------------------- | -------------------------------------------------------------------------- | ----------------------- | -| 🎯**احتياطي ذكي من 4 طبقات** | المسار التلقائي: الاشتراك → مفتاح API → رخيص → مجاني | -| 📊**تتبع الحصص في الوقت الفعلي** | عدد الرموز الحية + إعادة تعيين العد التنازلي لكل مزود | -| 🔄**تنسيق الترجمة** | OpenAI ↔ Claude ↔ Gemini ↔ الردود مع التحويلات الآمنة للمخطط | -| 👥**دعم الحسابات المتعددة** | حسابات متعددة لكل مزود مع اختيار ذكي | -| 🔄**تحديث تلقائي للرمز** | يتم تحديث رموز OAuth المميزة تلقائيًا من خلال إعادة المحاولة | -| 🎨**مجموعات مخصصة** | 9 استراتيجيات موازنة + التحكم في السلسلة الاحتياطية | -| 🌐**جهاز توجيه Wildcard** | `المزود/*` التوجيه الديناميكي | -| 🧠**التفكير في ضوابط الميزانية** | حدود التفكير المنطقي والتلقائي والمخصص والتكيفي | -| 🔀**الأسماء المستعارة للنماذج** | مدمج + اسم مستعار للنموذج المخصص وأمان الترحيل | -| ⚡**تدهور الخلفية** | قم بتوجيه مهام الخلفية ذات الأولوية المنخفضة إلى نماذج أرخص | -| 🧪**التوجيه الذكي المدرك للمهام** | تحديد النموذج تلقائيًا حسب نوع المحتوى (الترميز/الرؤية/التحليل/التلخيص) | -| 🔄**سير عمل وكيل A2A** | منسق ولايات ميكرونيزيا الموحدة الحتمية لعمليات إعدام الوكيل متعددة الخطوات | -| 🔀**التوجيه التكيفي** | تجاوز الإستراتيجية الديناميكية بناءً على حجم الرمز المميز والتعقيد الفوري | -| 🎲**تنوع مقدمي الخدمة** | شانون الإنتروبيا التهديف موازنة توزيع حركة المرور والسرد التلقائي | -| 💬**الحقن الفوري للنظام** | يتم تطبيق ضوابط السلوك العالمية بشكل متسق | -| 📄**توافق واجهة برمجة التطبيقات للردود** | الدعم الكامل `/v1/responses` لـ Codex وسير العمل الوكيل المتقدم | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| ميزة | ماذا يفعل | -| ------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ---------------------------------------- | -| 🖼️**إنشاء الصور** | `/v1/images/ Generations` مع الواجهات الخلفية السحابية والمحلية | -| 📐**المضامين** | `/v1/embeddings` لخطوط أنابيب البحث وRAG | -| 🎤**نسخ صوتي** | `/v1/audio/transcriptions` - 7 مقدمي خدمات (Deepgram Nova 3، AssemblyAI، Groq Whisper، HuggingFace، ElevenLabs، OpenAI، Azure)، الكشف التلقائي عن اللغة، دعم MP4/MP3/WAV | -| 🔊**تحويل النص إلى كلام** | `/v1/audio/speech` - 10 مقدمي خدمات (ElevenLabs، OpenAI، Deepgram، Cartesia، PlayHT، HuggingFace، Nvidia NIM، Inworld، Coqui، Tortoise) مع رسائل الخطأ الصحيحة | -| 🎬**توليد الفيديو** | `/v1/videos/أجيال` (سير عمل ComfyUI + SD WebUI) | -| 🎵**جيل الموسيقى** | `/v1/music/generations` (سير عمل ComfyUI) | -| 🛡️**اعتدالات** | `/v1/moderations` فحوصات السلامة | -| 🔀**إعادة الترتيب** | `/v1/rerank` لدرجات الملاءمة | -| 🔍**بحث الويب**🆕 | `/v1/search` - 5 مقدمي خدمات (Serper، Brave، Perplexity، Exa، Tavily)، أكثر من 6500 خدمة مجانية شهريًا، تجاوز الفشل التلقائي، ذاكرة التخزين المؤقت | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| ميزة | ماذا يفعل | -| --------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**قواطع الدائرة** | رحلة/استرداد لكل نموذج مع عناصر التحكم في العتبة | -| 🎯**نماذج تدرك نقطة النهاية** | تعلن النماذج المخصصة عن نقاط النهاية المدعومة + تنسيق API | -| 🛡️**القطيع المضاد للرعد** | حماية Mutex + الإشارة في أحداث إعادة المحاولة/التقييم | -| 🧠**ذاكرة التخزين المؤقت الدلالية + التوقيع** | تقليل التكلفة/زمن الوصول باستخدام طبقتين من ذاكرة التخزين المؤقت | -| ⚡**طلب العجز** | نافذة الحماية المكررة | -| 🔒**انتحال بصمة الإصبع TLS** | بصمة TLS الشبيهة بالمتصفح -**تقلل من اكتشاف الروبوتات ووضع علامة على الحساب** | -| 🔏**مطابقة بصمة CLI** | يطابق توقيعات طلب واجهة سطر الأوامر (CLI) الأصلية -**يقلل من مخاطر الحظر مع الحفاظ على عنوان IP الخاص بالوكيل** | -| 🌐**تصفية IP** | التحكم في القائمة المسموح بها/القائمة المحظورة لعمليات النشر المكشوفة | -| 📊**حدود المعدل القابلة للتحرير** | حدود عالمية/مستوى مزود قابلة للتكوين مع الثبات | -| 📉**التحلل الرشيق** | قدرات احتياطية متعددة الطبقات تحمي عمليات البوابة الأساسية | -| 📜**مسار تدقيق التكوين** | تتبع التغيير القائم على الاختلاف يمنع الانحراف التشغيلي من خلال عمليات التراجع البسيطة | -| ⏳**مزامنة صحة الموفر** | مراقبة استباقية لانتهاء صلاحية الرمز المميز، مما يؤدي إلى تنبيهات قبل فشل التفويض | -| 🚪**تعطيل الحسابات المحظورة تلقائيًا** | يقوم قاطع الدائرة التشغيلية بإغلاق حسابات الرموز المميزة المحظورة بشكل دائم تلقائيًا | -| 🔑**إدارة مفاتيح واجهة برمجة التطبيقات + تحديد النطاق** | تأمين إصدار/تدوير المفتاح وضوابط النموذج/المزود | -| 👁️**الكشف عن مفتاح واجهة برمجة التطبيقات (Scoped API)**🆕 | الاشتراك في استرداد مفاتيح واجهة برمجة التطبيقات عبر `ALLOW_API_KEY_REVEAL` | -| 🛡️**محميه `/موديلات`** | بوابة مصادقة اختيارية وإخفاء الموفر لكتالوج النماذج | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| ميزة | ماذا يفعل | -| ---------------------------------- | ------------------------------------------------------------------------- | ---------------------------- | -| 📝**الطلب + تسجيل الوكيل** | الطلب/الاستجابة الكاملة وتسجيل الوكيل | -| 📉**السجلات التفصيلية المتدفقة**🆕 | يعيد بناء تدفقات حمولة SSE بشكل واضح في واجهة المستخدم | -| 📋**لوحة تحكم السجلات الموحدة** | طلب العروض والوكيل والتدقيق ووحدة التحكم في صفحة واحدة | -| 🔍**طلب القياس عن بعد** | زمن الاستجابة p50/p95/p99 وطلب التتبع | -| 🏥**لوحة المعلومات الصحية** | وقت التشغيل، حالات الكسارة، عمليات الإغلاق، إحصائيات ذاكرة التخزين المؤقت | -| 💰**تتبع التكلفة** | ضوابط الميزانية ورؤية التسعير لكل نموذج | -| 📈**تصورات التحليلات** | رؤى استخدام النموذج/الموفر وطرق عرض الاتجاه | -| 🧪**إطار التقييم** | اختبار المجموعة الذهبية مع استراتيجيات المطابقة القابلة للتكوين | -| 📡**تشخيص مباشر**🆕 | تجاوز ذاكرة التخزين المؤقت الدلالية لإجراء اختبار مباشر دقيق للسرد | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| ميزة | ماذا يفعل | -| --------------------------------------------- | -------------------------------------------------------------------- | --------------------- | -| 🌐**النشر في أي مكان** | المضيف المحلي، VPS، Docker، البيئات السحابية | -| 🚇**نفق كلاود فلير**🆕 | تكامل النفق السريع بنقرة واحدة من لوحة المعلومات | -| 🔑**تصفية نموذج مفتاح واجهة برمجة التطبيقات** | تمت تصفية الاستجابة الأصلية /v1/models عبر أدوار سياق الحامل المعينة | -| ⚡**تجاوز ذاكرة التخزين المؤقت الذكية** | استدلالات TTL قابلة للتكوين وضوابط إعادة الجلب القسري | -| 🔄**النسخ الاحتياطي/الاستعادة** | تدفقات التصدير/الاستيراد والتعافي من الكوارث | -| 🧙**معالج الإعداد** | الإعداد الموجه لأول مرة | -| 🔧**لوحة تحكم أدوات CLI** | إعداد بنقرة واحدة لأدوات الترميز الشائعة | -| 🎮**ساحة اللعب النموذجية** | اختبر أي موفر/نموذج/نقطة نهاية من لوحة المعلومات | -| 🔏**تبديل بصمة الإصبع CLI** | مطابقة بصمات الأصابع لكل موفر في الإعدادات > الأمان | -| 🌐**i18n (30 لغة)** | لوحة تحكم كاملة + دعم لغة المستندات مع تغطية RTL | -| 🧹**مسح كافة النماذج** | مسح قائمة النماذج بنقرة واحدة في تفاصيل المزود | -| 👁️**عناصر التحكم في الشريط الجانبي**🆕 | إخفاء المكونات وعمليات التكامل من إعدادات المظهر | -| 📋**نماذج الإصدارات** | قوالب GitHub الموحدة للأخطاء والميزات | -| 📂**دليل البيانات المخصصة** | تجاوز `DATA_DIR` لموقع التخزين | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1294,105 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -عند فشل الحصة أو المعدل أو الصحة، ينتقل OmniRoute تلقائيًا إلى المرشح التالي دون التبديل اليدوي.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- يمكن اكتشاف MCP + A2A في واجهة المستخدم والمستندات (غير مخفية) -- تعرض واجهات برمجة التطبيقات لحالة البروتوكول البيانات التشغيلية المباشرة (`/api/mcp/*`، `/api/a2a/*`) -- تتضمن لوحات المعلومات إجراءات لعمليات اليوم الثاني (تبديل التحرير والسرد، وإعادة ضبط الكسارة، وإلغاء المهام)#### Translator + validation workflow +#### Protocol management that is visible and operable -منطقة المترجم تشمل: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**الملعب**: طلب عمليات التحقق من التحويل -**أداة اختبار الدردشة**: الطلب/الإجابة الكاملة ذهابًا وإيابًا -**منصة الاختبار**: حالات متعددة في جولة واحدة -**المراقبة المباشرة**: عرض حركة المرور في الوقت الحقيقي +#### Translator + validation workflow -بالإضافة إلى التحقق من صحة البروتوكول مع عملاء حقيقيين عبر اختبار تشغيل npm:البروتوكولات:e2e. +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— مرجع الأداة، وتكوينات IDE، وأمثلة العميل +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A Server README](src/lib/a2a/README.md)**— المهارات، وأساليب JSON-RPC، والبث، ودورة حياة المهمة## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -يشتمل OmniRoute على إطار تقييم مدمج لاختبار جودة استجابة LLM مقابل المجموعة الذهبية. يمكنك الوصول إليه عبر**Analytics → Evals**في لوحة التحكم.### Built-in Golden Set +## 🧪 Evaluations (Evals) -تحتوي "OmniRoute Golden Set" المحملة مسبقًا على حالات اختبار لما يلي: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- تحياتي، الرياضيات، الجغرافيا، توليد التعليمات البرمجية -- الامتثال لتنسيق JSON والترجمة وإنشاء تخفيض السعر -- رفض السلامة (المحتوى الضار)، العد، المنطق المنطقي### Evaluation Strategies +### Built-in Golden Set -| استراتيجية | الوصف | مثال | -| ---------------- | ------------------------------------------------------------- | --------------------------------- | --- | -| `بالضبط` | يجب أن يتطابق الإخراج تمامًا مع | `"4"` | -| `يحتوي على` | يجب أن يحتوي الإخراج على سلسلة فرعية (غير حساسة لحالة الأحرف) | `"باريس"` | -| "التعبير العادي" | يجب أن يتطابق الإخراج مع نمط regex | `"1.*2.*3"` | -| "مخصص" | ترجع دالة JS المخصصة صواب/خطأ | `(الإخراج) => الإخراج.الطول > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) -<التفاصيل> +
+🧩 MCP Setup (Model Context Protocol) -🧩 إعداد MCP (بروتوكول السياق النموذجي) +Start MCP transport in stdio mode: -بدء نقل MCP في وضع stdio:```bash +```bash omniroute --mcp +``` -```` +Recommended validation flow: -تدفق التحقق الموصى به: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. قم بتوصيل عميل MCP الخاص بك عبر stdio. -2. قم بتشغيل "omniroute_get_health". -3. قم بتشغيل "omniroute_list_combos". -4. افتح `/dashboard/mcp` لتأكيد نبضات القلب والنشاط والتدقيق. +Useful APIs for automation: -واجهات برمجة التطبيقات المفيدة للأتمتة: +- `GET /api/mcp/status` +- `GET /api/mcp/tools` +- `GET /api/mcp/audit` +- `GET /api/mcp/audit/stats` -- `الحصول على /api/mcp/status` -- `الحصول على /api/mcp/tools` -- `الحصول على /api/mcp/audit` -- `الحصول على /api/mcp/audit/stats`
+ -<التفاصيل> -🤝 إعداد A2A (Agent2Agent) +
+🤝 A2A Setup (Agent2Agent) -اكتشف الوكيل:```bash +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -إرسال مهمة:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` +Manage lifecycle: -إدارة دورة الحياة: - -- `الحصول على /api/a2a/status` -- `الحصول على /api/a2a/tasks` -- `الحصول على /api/a2a/tasks/:id` +- `GET /api/a2a/status` +- `GET /api/a2a/tasks` +- `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -واجهة المستخدم التشغيلية: +Operational UI: -- `/dashboard/a2a` لإمكانية ملاحظة المهمة/الحالة/الدفق وإجراءات الدخان
+- `/dashboard/a2a` for task/state/stream observability and smoke actions -<التفاصيل> -🧪 التحقق من صحة البروتوكول الشامل + -التحقق من صحة كلا البروتوكولين مع عملاء حقيقيين:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -هذا يتحقق: +This verifies: -- اتصال/قائمة/اتصال عميل MCP SDK -- اكتشاف A2A/إرسال/دفق/حصول على/إلغاء -- التحقق من البيانات في تدقيق MCP وواجهات برمجة التطبيقات لإدارة المهام A2A
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs -<التفاصيل> + -💳 مقدمو الاشتراكات### Claude Code (Pro/Max) +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1405,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**نصيحة احترافية:**استخدم Opus للمهام المعقدة، وSonnet للسرعة. OmniRoute يتتبع الحصة لكل نموذج!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1419,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -يحتوي كل حساب Codex الآن على تبديل السياسة في "لوحة المعلومات -> مقدمي الخدمة": +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (تشغيل/إيقاف): فرض سياسة عتبة النافذة البالغة 5 ساعات. -- `أسبوعيًا` (تشغيل/إيقاف): فرض سياسة حد النافذة الأسبوعية. -- سلوك العتبة: عندما تصل النافذة الممكّنة إلى >=90% من الاستخدام، يتم تخطي هذا الحساب. -- سلوك التناوب: يقوم OmniRoute بتوجيه حساب Codex المؤهل التالي تلقائيًا. -- إعادة تعيين السلوك: عندما يمر وقت الموفر `resetAt`، يصبح الحساب مؤهلاً مرة أخرى تلقائيًا. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -السيناريوهات: +Scenarios: -- `5h ON' + `Weekly ON`: يتم تخطي الحساب عندما تصل أي من النافذتين إلى الحد الأدنى. -- `إيقاف لمدة 5 ساعات` + `تشغيل أسبوعي`: الاستخدام الأسبوعي فقط يمكنه حظر الحساب. -- `5 ساعات تشغيل' + `إيقاف أسبوعي`: الاستخدام لمدة 5 ساعات فقط يمكنه حظر الحساب. -- تم `resetAt`: يعود الحساب إلى التدوير تلقائيًا (لا توجد إعادة تمكين يدوية).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1444,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**أفضل قيمة:**طبقة مجانية ضخمة! استخدم هذا قبل المستويات المدفوعة.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1459,74 +1662,91 @@ Models:
-<التفاصيل> +
+🔑 API Key Providers -🔑 موفري مفاتيح واجهة برمجة التطبيقات### NVIDIA NIM (FREE developer access — 70+ models) +### NVIDIA NIM (FREE developer access — 70+ models) -1. قم بالتسجيل: [build.nvidia.com](https://build.nvidia.com) -2. احصل على مفتاح واجهة برمجة التطبيقات (API) مجانًا (يتضمن 1000 نقطة استدلال) -3. لوحة المعلومات → إضافة موفر → NVIDIA NIM: - - مفتاح API: `nvapi-your-key` +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**النماذج:**`nvidia/llama-3.3-70b-instruct`، `nvidia/mistral-7b-instruct`، وأكثر من 50 طرازًا آخر +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -**نصيحة احترافية:**واجهة برمجة التطبيقات المتوافقة مع OpenAI — تعمل بسلاسة مع ترجمة تنسيق OmniRoute!### DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -1. قم بالتسجيل: [platform.deepseek.com](https://platform.deepseek.com) -2. احصل على مفتاح API -3. لوحة المعلومات → إضافة موفر → DeepSeek +### DeepSeek -**النماذج:**`deepseek/deepseek-chat`، `deepseek/deepseek-coder`### Groq (Free Tier Available!) +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -1. قم بالتسجيل: [console.groq.com](https://console.groq.com) -2. احصل على مفتاح API (الطبقة المجانية متضمنة) -3. لوحة المعلومات → إضافة موفر → Groq +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**النماذج:**`groq/llama-3.3-70b`، `groq/mixtral-8x7b` +### Groq (Free Tier Available!) -**نصيحة احترافية:**استنتاج فائق السرعة — الأفضل للبرمجة في الوقت الفعلي!### OpenRouter (100+ Models) +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -1. قم بالتسجيل: [openrouter.ai](https://openrouter.ai) -2. احصل على مفتاح API -3. لوحة المعلومات → إضافة موفر → OpenRouter +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**النماذج:**يمكنك الوصول إلى أكثر من 100 نموذج من جميع المزودين الرئيسيين من خلال مفتاح واجهة برمجة التطبيقات (API) واحد. +**Pro Tip:** Ultra-fast inference — best for real-time coding! -**سلوك لوحة المعلومات:**تتم إدارة نماذج OpenRouter من**النماذج المتوفرة**. تعمل عمليات الإضافة والاستيراد والمزامنة التلقائية يدويًا على تحديث نفس القائمة.
+### OpenRouter (100+ Models) -<التفاصيل> +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -💰 مقدمو الخدمة الرخيصة (النسخ الاحتياطي)### GLM-4.7 (Daily reset, $0.6/1M) +**Models:** Access 100+ models from all major providers through a single API key. -1. قم بالتسجيل: [Zhipu AI](https://open.bigmodel.cn/) -2. احصل على مفتاح API من خطة الترميز -3. لوحة المعلومات → إضافة مفتاح واجهة برمجة التطبيقات: - - المزود: `glm` - - مفتاح واجهة برمجة التطبيقات: "مفتاحك". +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -**الاستخدام:**`glm/glm-4.7` + -**نصيحة احترافية:**توفر خطة البرمجة حصة 3× بتكلفة 1/7! إعادة الضبط يوميًا الساعة 10:00 صباحًا.### MiniMax M2.1 (5h reset, $0.20/1M) +
+💰 Cheap Providers (Backup) -1. قم بالتسجيل: [MiniMax](https://www.minimax.io/) -2. احصل على مفتاح API -3. لوحة المعلومات → إضافة مفتاح API +### GLM-4.7 (Daily reset, $0.6/1M) -**الاستخدام:**`minimax/MiniMax-M2.1` +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**نصيحة احترافية:**الخيار الأرخص للسياق الطويل (مليون رمز)!### Kimi K2 ($9/month flat) +**Use:** `glm/glm-4.7` -1. اشترك: [Moonshot AI](https://platform.moonshot.ai/) -2. احصل على مفتاح API -3. لوحة المعلومات → إضافة مفتاح API +**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. -**الاستخدام:**`كيمي/كيمي-أحدث` +### MiniMax M2.1 (5h reset, $0.20/1M) -**نصيحة احترافية:**سعر ثابت قدره 9 دولارات شهريًا مقابل 10 ملايين رمز مميز = 0.90 دولارًا أمريكيًا/مليون تكلفة فعالة!
+1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key -<التفاصيل> +**Use:** `minimax/MiniMax-M2.1` -🆓 مقدمو الخدمة مجانًا (النسخ الاحتياطي في حالات الطوارئ)### Qoder (5 FREE models via OAuth) +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1567,9 +1787,10 @@ Models:
-<التفاصيل> +
+🎨 Create Combos -🎨 أنشئ مجموعات### Example 1: Maximize Subscription → Cheap Backup +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1597,9 +1818,10 @@ Cost: $0 forever!
-<التفاصيل> +
+🔧 CLI Integration -🔧 تكامل CLI### Cursor IDE +### Cursor IDE ``` Settings → Models → Advanced: @@ -1610,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -استخدم صفحة**أدوات CLI**في لوحة المعلومات للتكوين بنقرة واحدة، أو قم بتحرير `~/.claude/settings.json` يدويًا.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1621,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**الخيار 1 — لوحة التحكم (مستحسن):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**الخيار 2 - يدويًا:**تحرير `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1638,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **ملاحظة:**يعمل OpenClaw فقط مع OmniRoute المحلي. استخدم "127.0.0.1" بدلاً من "المضيف المحلي" لتجنب مشكلات دقة IPv6.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1652,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**الخطوة 1:**أضف OmniRoute كموفر مخصص:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**الخطوة 2:**إنشاء/تحرير `opencode.json` في جذر مشروعك:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1678,117 +1909,130 @@ opencode } } } -```` +``` -**الخطوة 3:**حدد النموذج في OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**نصيحة:**أضف أي نموذج متوفر في نقطة نهاية OmniRoute `/v1/models` إلى قسم `models`. استخدم التنسيق "provider/model-id" من لوحة معلومات OmniRoute.
+ --- ## استكشاف الأخطاء -<التفاصيل> -انقر لتوسيع دليل استكشاف الأخطاء وإصلاحها +
+Click to expand troubleshooting guide -**"نموذج اللغة لم يقدم رسائل"** +**"Language model did not provide messages"** -- استنفدت حصة الموفر → تحقق من تعقب حصة الموفر في لوحة المعلومات -- الحل: استخدم خيار التحرير والسرد الاحتياطي أو قم بالتبديل إلى مستوى أرخص +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**الحد من المعدل** +**Rate limiting** -- حصة الاشتراك المحددة → الرجوع إلى GLM/MiniMax -- إضافة التحرير والسرد: `cc/clude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**انتهت صلاحية رمز OAuth** +**OAuth token expired** -- يتم التحديث تلقائيًا بواسطة OmniRoute -- إذا استمرت المشكلات: لوحة المعلومات → الموفر → إعادة الاتصال +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**تكاليف مرتفعة** +**High costs** -- التحقق من إحصائيات الاستخدام في لوحة المعلومات → التكاليف -- تبديل النموذج الأساسي إلى GLM/MiniMax -- استخدم الطبقة المجانية (Gemini CLI، Qoder) للمهام غير الحرجة +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**منافذ لوحة المعلومات/واجهة برمجة التطبيقات غير صحيحة** +**Dashboard/API ports are wrong** -- `PORT` هو المنفذ الأساسي الأساسي (ومنفذ API افتراضيًا) -- `API_PORT` يتخطى فقط مستمع واجهة برمجة التطبيقات المتوافق مع OpenAI -- `DASHBOARD_PORT` يتجاوز مستمع لوحة المعلومات/Next.js فقط -- قم بتعيين `NEXT_PUBLIC_BASE_URL` على لوحة التحكم/عنوان URL العام (لردود اتصال OAuth) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**أخطاء المزامنة السحابية** +**Cloud sync errors** -- تحقق من نقاط `BASE_URL` لمثيلك قيد التشغيل -- تحقق من نقاط `CLOUD_URL` إلى نقطة النهاية السحابية المتوقعة -- حافظ على محاذاة قيم `NEXT_PUBLIC_*` مع القيم الموجودة على جانب الخادم +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**تسجيل الدخول الأول لا يعمل** +**First login not working** -- حدد "INITIAL_PASSWORD" في ".env". -- في حالة عدم تعيينها، تكون كلمة المرور الاحتياطية هي `123456` +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**لا توجد سجلات الطلب** +**No request logs** -- تتم كتابة عناصر الطلب إلى `DATA_DIR/call_logs/` كملف JSON واحد لكل طلب -- تمكين التقاط خط الأنابيب من لوحة المعلومات → السجلات → طلب السجلات إذا كنت بحاجة إلى حمولات مفصلة لكل مرحلة -- اضبط `APP_LOG_TO_FILE=true` إذا كنت تريد أيضًا وجود سجلات لوحدة تحكم التطبيق في `logs/application/app.log` -- اضبط `APP_LOG_MAX_FILE_SIZE`، و`APP_LOG_RETENTION_DAYS`، و`APP_LOG_MAX_FILES`، و`CALL_LOG_MAX_ENTRIES` حسب الحاجة +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**يظهر اختبار الاتصال "غير صالح" لمقدمي الخدمات المتوافقين مع OpenAI** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- لا يكشف العديد من مقدمي الخدمة عن نقطة نهاية `/models` -- يتضمن OmniRoute v1.0.6+ التحقق الاحتياطي من خلال إكمال الدردشة -- تأكد من أن عنوان URL الأساسي يتضمن لاحقة `/v1`### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ مهم للمستخدمين الذين يقومون بتشغيل OmniRoute على VPS أو Docker أو أي خادم بعيد**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -يستخدم موفرو**Antigravity**و**Gemini CLI****Google OAuth 2.0**. تتطلب Google أن يكون `redirect_uri` في تدفق OAuth مطابقًا تمامًا لأحد معرفات URI المسجلة مسبقًا في Google Cloud Console للتطبيق. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -يتم تسجيل بيانات اعتماد OAuth المجمعة في OmniRoute**لـ `المضيف المحلي` فقط**. عند الوصول إلى OmniRoute على خادم بعيد (على سبيل المثال، `https://omniroute.myserver.com`)، يرفض Google المصادقة باستخدام:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -يلزمك إنشاء**OAuth 2.0 Client ID**في Google Cloud Console باستخدام معرف URI الخاص بخادمك.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. افتح Google Cloud Console** +#### Step-by-step -انتقل إلى: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. قم بإنشاء معرف عميل OAuth 2.0 جديد** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- انقر على**"+ إنشاء بيانات اعتماد"**→**"معرف عميل OAuth"** -- نوع التطبيق:**"تطبيق ويب"** -- الاسم: أي شيء تريده (على سبيل المثال، "OmniRoute Remote") +**2. Create a new OAuth 2.0 Client ID** -**3. أضف عناوين URI لإعادة التوجيه المعتمدة** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -في الحقل**"عناوين URI لإعادة التوجيه المعتمدة"**، أضف:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> استبدل "your-server.com" بنطاق الخادم الخاص بك أو عنوان IP (قم بتضمين المنفذ إذا لزم الأمر، على سبيل المثال "http://45.33.32.156:20128/callback"). +**4. Save and copy the credentials** -**4. حفظ ونسخ بيانات الاعتماد** +After creating, Google will show the **Client ID** and **Client Secret**. -بعد الإنشاء، ستعرض Google**معرف العميل**و**سر العميل**. +**5. Set environment variables** -**5. تعيين متغيرات البيئة** +In your `.env` (or Docker environment variables): -في `.env` (أو متغيرات بيئة Docker):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1797,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. أعد تشغيل OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. حاول الاتصال مرة أخرى** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -لوحة المعلومات → الموفرون → Antigravity (أو Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -سيقوم Google الآن بإعادة التوجيه بشكل صحيح إلى `https://your-server.com/callback`.--- +--- #### Temporary workaround (without custom credentials) -إذا كنت لا ترغب في إعداد بيانات الاعتماد الخاصة بك الآن، فلا يزال بإمكانك استخدام**تدفق عنوان URL اليدوي**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. يفتح OmniRoute عنوان URL لتفويض Google -2. بعد التفويض، يحاول Google إعادة التوجيه إلى "المضيف المحلي" (والذي يفشل على الخادم البعيد) -3.**انسخ عنوان URL الكامل**من شريط عنوان المتصفح (حتى لو لم يتم تحميل الصفحة) -4. الصق عنوان URL هذا في الحقل الموضح في نموذج اتصال OmniRoute -5. انقر**"اتصال"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> يعمل هذا لأن رمز التفويض الموجود في عنوان URL صالح بغض النظر عما إذا تم تحميل صفحة إعادة التوجيه أم لا.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. -<التفاصيل> -🇧🇷 النسخة البرتغالية#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -تم إثبات**Antigravity**و**Gemini CLI**باستخدام**Google OAuth 2.0**للمصادقة. تطلب Google أن يتم استخدام `redirect_uri` دون تدفق OAuth**بالتأكيد**إلى عناوين URI المسبقة لتطبيق Google Cloud Console. +
+🇧🇷 Versão em Português -نظرًا لأن اعتمادات OAuth المُدخلة ليست في OmniRoute، فهي عبارة عن سجلات**apenas لـ `المضيف المحلي`**. عند الوصول إلى OmniRoute من خادم بعيد (على سبيل المثال: `https://omniroute.meuservidor.com`)، أو تحصل Google على مصادقة عبر:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -يجب عليك إنشاء**OAuth 2.0 Client ID**على Google Cloud Console باستخدام URI لخادمك.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. الوصول إلى Google Cloud Console** +#### Passo a passo -العبرة: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Acesse o Google Cloud Console** -**2. طلب معرف عميل OAuth 2.0** +Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- انقر على**"+ إنشاء بيانات الاعتماد"**→**"معرف عميل OAuth"** -- نوع التطبيق:**"تطبيق ويب"** -- الاسم: escolha qualquer nome (على سبيل المثال: `OmniRoute Remote`) +**2. Crie um novo OAuth 2.0 Client ID** -**3. Adicione كمحددات URI لإعادة التوجيه المعتمدة** +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -ليس هناك مجال**"عناوين URI لإعادة التوجيه المعتمدة"**، أضف:``` +**3. Adicione as Authorized Redirect URIs** + +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> استبدل `seu-servidor.com` بمنطقتك أو IP بخادمك (بما في ذلك البوابة إذا لزم الأمر، على سبيل المثال: `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. حفظ ونسخ كموثقات** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -وبعد ذلك، قم بإنشاء أو عرض Google o**معرف العميل**أو**سر العميل**. +**5. Configure as variáveis de ambiente** -**5. تكوين كمتغيرات البيئة** +No seu `.env` (ou nas variáveis de ambiente do Docker): -ليس لديك `.env` (أو في بيئة Docker المتنوعة):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1876,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reinicie أو OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute +``` -```` +**7. Tente conectar novamente** -**7. خيمة تواصل جديدة** +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -لوحة المعلومات → الموفرون → Antigravity (ou Gemini CLI) → OAuth +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. -قم بإعادة توجيه Google بشكل صحيح إلى `https://seu-servidor.com/callback` ووظيفة المصادقة.--- +--- #### Workaround temporário (sem configurar credenciais próprias) -إذا لم ترغب في إنشاء بيانات اعتماد خاصة بك منذ الآن، فمن الممكن استخدام التدفق**دليل URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. يفتح OmniRoute عنوان URL لتفويض Google -2. نسمح لك بأن تقوم Google بإعادة التوجيه إلى "المضيف المحلي" (الذي لا يوجد خادم عن بعد) -3.**انسخ عنوان URL كاملاً**من شريط الإدخال في متصفحك (حتى لا يتم نقل الصفحة) -4. هذا هو عنوان URL الذي يظهر في وضع الاتصال بـ OmniRoute -5. انقر على**"الاتصال"** +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) +4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute +5. Clique em **"Connect"** -> يعمل هذا الحل البديل لأن رمز التفويض الموجود على عنوان URL يكون صالحًا بشكل مستقل لإعادة التوجيه حيث يتم تحميله أو لا.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1914,64 +2171,73 @@ docker restart omniroute ## 🛠️ Tech Stack -<التفاصيل> -انقر لتوسيع تفاصيل المجموعة التقنية +
+Click to expand tech stack details --**وقت التشغيل**: Node.js 18–22 LTS (⚠️ Node.js 24+**غير مدعومة**— الثنائيات الأصلية `better-sqlite3` غير متوافقة) --**اللغة**: TypeScript 5.9 —**TypeScript بنسبة 100%**عبر `src/` و`open-sse/` (لا يوجد `any` في الوحدات الأساسية منذ الإصدار 2.0) --**الإطار**: Next.js 16 + React 19 + Tailwind CSS 4 --**قاعدة البيانات**: LowDB (JSON) + SQLite (حالة المجال + سجلات الوكيل + تدقيق MCP + قرارات التوجيه) --**المخططات**: Zod (التحقق من صحة الإدخال/الإخراج لأداة MCP، وعقود API) --**البروتوكولات**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**البث**: الأحداث المرسلة من الخادم (SSE) --**المصادقة**: OAuth 2.0 (PKCE) + JWT + مفاتيح API + ترخيص نطاق MCP --**الاختبار**: مشغل اختبار Node.js + Vitest (أكثر من 900 اختبار بما في ذلك الوحدة والتكامل وE2E) --**CI/CD**: إجراءات GitHub (نشر npm التلقائي + Docker Hub عند الإصدار) --**الموقع الإلكتروني**: [omniroute.online](https://omniroute.online) --**الحزمة**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**دوكر**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**المرونة**: قاطع الدائرة، والتراجع الأسي، وقطيع مكافحة الرعد، وانتحال TLS، والإصلاح الذاتي للتحرير والسرد التلقائي
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## التوثيق -| وثيقة | الوصف | +| Document | Description | | ---------------------------------------------- | --------------------------------------------------- | -| [دليل المستخدم](docs/USER_GUIDE.md) | مقدمو الخدمات، والمجموعات، وتكامل CLI، والنشر | -| [مرجع واجهة برمجة التطبيقات](docs/API_REFERENCE.md) | جميع نقاط النهاية مع الأمثلة | -| [خادم MCP](open-sse/mcp-server/README.md) | 16 أدوات MCP وتكوينات IDE وعملاء Python/TS/Go | -| [خادم A2A](src/lib/a2a/README.md) | بروتوكول JSON-RPC 2.0، المهارات، التدفق، إدارة المهام | -| [محرك التحرير والسرد التلقائي](docs/auto-combo.md) | تسجيل 6 عوامل، حزم الوضع، الشفاء الذاتي | -| [استكشاف الأخطاء وإصلاحها](docs/TROUBLESHOOTING.md) | المشاكل والحلول الشائعة | -| [هندسة معمارية](docs/ARCHITECTURE.md) | بنية النظام والداخلية | -| [مساهمة](CONTRIBUTING.md) | إعداد التطوير والمبادئ التوجيهية | -| [مواصفات OpenAPI](docs/openapi.yaml) | مواصفات OpenAPI 3.0 | -| [سياسة الأمان](SECURITY.md) | الإبلاغ عن الثغرات الأمنية والممارسات الأمنية | -| [نشر الجهاز الافتراضي](docs/VM_DEPLOYMENT_GUIDE.md) | الدليل الكامل: إعداد VM + nginx + Cloudflare | -| [معرض الميزات](docs/FEATURES.md) | جولة لوحة القيادة المرئية مع لقطات الشاشة | -| [قائمة مراجعة الإصدار](docs/RELEASE_CHECKLIST.md) | خطوات التحقق من صحة الإصدار المسبق |--- +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -يحتوي OmniRoute على**210+ ميزات مخطط لها**عبر مراحل تطوير متعددة. فيما يلي المجالات الرئيسية: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| الفئة | الميزات المخططة | أبرز الأحداث | +| Category | Planned Features | Highlights | | ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | -| 🧠**التوجيه والاستخبارات**| 25+ | التوجيه ذو زمن الاستجابة الأقل، والتوجيه القائم على العلامات، والاختبار المبدئي للحصة، واختيار حساب P2C | -| 🔒**الأمان والامتثال**| 20+ | تقوية SSRF، وإخفاء بيانات الاعتماد، والحد الأقصى للمعدل لكل نقطة نهاية، وتحديد نطاق مفتاح الإدارة | -| 📊**قابلية الملاحظة**| 15+ | تكامل OpenTelemetry ومراقبة الحصص في الوقت الفعلي وتتبع التكلفة لكل نموذج | -| 🔄**تكامل الموفر**| 20+ | تسجيل النموذج الديناميكي، فترات تهدئة الموفر، الدستور الغذائي متعدد الحسابات، تحليل حصة الطيار المساعد | -| ⚡**الأداء**| 15+ | طبقة ذاكرة التخزين المؤقت المزدوجة، ذاكرة التخزين المؤقت السريعة، ذاكرة التخزين المؤقت للاستجابة، استمرار البث، واجهة برمجة التطبيقات الدفعية | -| 🌐**النظام البيئي**| 10+ | WebSocket API، إعادة تحميل التكوين السريع، مخزن التكوين الموزع، الوضع التجاري |### 🔜 Coming Soon +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**تكامل OpenCode**— دعم الموفر الأصلي لـ OpenCode AI IDE للترميز -- 🔗**تكامل TRAE**— الدعم الكامل لإطار تطوير TRAE AI -- 📦**Batch API**— معالجة الدفعات غير المتزامنة للطلبات المجمعة -- 🎯**التوجيه المعتمد على العلامات**— توجيه الطلبات بناءً على العلامات المخصصة والبيانات الوصفية -- 💰**إستراتيجية أقل تكلفة**— تحديد أرخص مزود متاح تلقائيًا +### 🔜 Coming Soon -> 📝 مواصفات الميزات الكاملة متوفرة في [`docs/new-features/`](docs/new-features/) (217 مواصفات تفصيلية)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1979,18 +2245,20 @@ docker restart omniroute ### How to Contribute -1. شوكة المستودع -2. قم بإنشاء فرع الميزات الخاص بك (`git checkout -b feature/amazing-feature`) -3. تنفيذ التغييرات ("git الالتزام -m "إضافة ميزة مذهلة") -4. ادفع إلى الفرع ("ميزة git Push Origin/ميزة مذهلة") -5. افتح طلب السحب +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -راجع [CONTRIBUTING.md](CONTRIBUTING.md) للحصول على إرشادات مفصلة.### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -2002,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -شكر خاص لـ**[9router](https://github.com/decolua/9router)**بواسطة**[decolua](https://github.com/decolua)**— المشروع الأصلي الذي ألهم هذه الشوكة. يعتمد OmniRoute على هذا الأساس المذهل مع ميزات إضافية وواجهات برمجة التطبيقات متعددة الوسائط وإعادة كتابة TypeScript كاملة. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -شكر خاص لـ**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— تطبيق Go الأصلي الذي ألهم منفذ JavaScript هذا.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## الرخصة -ترخيص MIT - راجع [الترخيص](الترخيص) للحصول على التفاصيل.--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/ar/docs/ARCHITECTURE.md b/docs/i18n/ar/docs/ARCHITECTURE.md index fb5c25ffba..ec6712920b 100644 --- a/docs/i18n/ar/docs/ARCHITECTURE.md +++ b/docs/i18n/ar/docs/ARCHITECTURE.md @@ -4,257 +4,291 @@ --- -_آخر تحديث: 2026-03-28_## الملخص التنفيذي -OmniRoute عبارة عن بوابة توجيه نقطة تعمل بالذكاء الاصطناعي ولوحة معلومات مبنية على Next.js. -وهو يوفر نقطة نهاية واحدة متوافقة مع OpenAI (`/v1/*`) ويوجه حركة المرور عبر العديد من الخدمات الموفري الأولية مع الترجمة والاحتياط وتحديث الرمز المميز وتتبع الاستخدام. -التان الأساسية: +_Last updated: 2026-03-28_ -- سطح API متوافق مع OpenAI لـ CLI/الأدوات (28 منتجًا) -- ترجمة الطلب/الاستجابة عبر التنسيقات الموفر -- نموذج بناء التحرير والسرد (سلسلة الارتباطات المتعددة) -- موازنة حساب الحساب (حسابات متعددة لكل شخص) -- إدارة اتصال موفر OAuth + API-key -- إنشاء التضمين عبر `/v1/embeddings` (6 مقدمي خدمات، 9 نماذج) -- إنشاء الصور عبر `/v1/images/Generation` (4 مقدمي خدمات، 9 نماذج) -- فكر في تحليل العلامات (`...`) لنماذج الاستدلال -- تحديد القيمة للتوافق مع OpenAI SDK -- تطبيع الدور (المطور → النظام، النظام → المستخدم) للتوافق بين الموفرين -- تحويل المنتج منظم (json_schema → Gemini ResponseSchema) -- الثبات المحلي لمقدمي الخدمات والمفاتيح والأسماء المستعارة والمجموعات والإعدادات والتسعير -- تتبع تكلفة/التكلفة وتسجيل الطلب -- نوبات سحابية اختيارية للأجهزة/الحالة الثابتة -- القائمة الخاصة بها/القائمة المحظورة لـ IP للتحكم في الوصول إلى واجهة برمجة التطبيقات -- التفكير في إدارة الميزانية (العبور / التلقائي / المقصود / التكيفي) -- هيكل البناء العالمي -- تتبع البصمات -- تحديد المحسن لكل حساب مع الملفات الشخصية الخاصة بالمزود -- تقطع فاصل لمرونة المورد -- حماية القطيع ضد الرعد مع موتكس -- ذاكرة التخزين المؤقتة لإلغاء البيانات المكررة للطلبة المستندية للتوقيع -- المجال: توفر النموذج، وقواعد التكلفة، والسياسة الاحتياطية، وسياسة فك الضغط -- فرانسيسكوية المجال المجال (ذاكرة التخزين المؤقتة للكتاب في SQLite للاحتياطيات والميزانيات وفتح قواطع الضوء) -- السياسة التي تحدد الطلب المركزي (التأمين → الميزانية → الاحتياطي) -- طلب القياس عن بعد مع تجميع الكمون ص50/ص95/ص99 -- معرف الارتباط (X-Request-Id) للتتبع الشامل -- تسجيل تدقيق كامل مع إلغاء الاشتراك لمفتاح API -- إطار تقييمي وجودة LLM -- لوحة تحكم واجهة المستخدم المرنة مع فاصل زمني في العمل -- مفري OAuth المطاطيون (12 وحدة ضمن `src/lib/oauth/providers/`) +## Executive Summary -وقت نموذج التشغيل الأساسي: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- تقوم مسارات تطبيق Next.js ضمن `src/app/api/*` ولتتمكن كل من واجهات تطبيقات برمجة لوحة المعلومات وواجهات برمجة تطبيقات التوافق -- نواة توجيه/SSE اشترك في `src/sse/*` + `open-sse/*` تمويل مع تنفيذ الموفر والترجمة والتدفق والرجوع والاستخدام## النطاق والحدود### In Scope +Core capabilities: -- وقت تشغيل البوابة المحلية -- واجهات برمجة التطبيقات المبتكرة للوحة المعلومات -- مصادقة الموفر وتحديث الرمز المميز -- طلب الترجمة و التدفق SSE -- الحالة المحلية + استمرارية الاستخدام -- نوبات سحابية اختيارية### خارج النطاق +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) -- تنفيذ خدمة السحابية خلف `NEXT_PUBLIC_CLOUD_URL` -- مستوى تحرير السودان/مستوى التحكم خارج نطاق العمل -- ثنائيات CLI الخارجية نفسها (Claude CLI، Codex CLI، وما إلى ذلك) ## سطح لوحة القيادة (الحالي) +Primary runtime model: -الصفحة الرئيسية ضمن `src/app/(dashboard)/dashboard/`: +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage -- `/dashboard` - بداية سريعة + نظرة عامة على الموفر -- `/dashboard/endpoint` - وكيل نقطة النهاية + علامات نهاية نقطة النهاية MCP + A2A + API -- `/dashboard/providers` - اتصالات الموفر وبيانات الاعتماد -- `/dashboard/combos` - إستراتيجيات التحرير والسرد والقوالب وقواعد توجيه التطورات -- `/dashboard/costs` - تجميع الأسعار ورؤية الأسعار -- `/dashboard/analytics` - تحليلات تعاطيات البناء -- `/dashboard/limits` - ضوابط الحصص/المعدلات -- `/dashboard/cli-tools` - إعداد واجهة سطر مودم، والكشف عن وقت التشغيل، ويشمل ذلك -- `/dashboard/agents` — تم ابتكار عملاء ACP + تسجيل عميل مخصص -- `/dashboard/media` — ساحة لعب الصور/الفيديو/الموسيقى -- `/dashboard/search-tools` - اختبار خريطة البحث -- `/dashboard/health` - وقت التشغيل، قواطع الدائرة، حدود المعدل -- `/dashboard/logs` - سجلات الطلب/الوكيل/التدقيق/وحدة التحكم -- `/dashboard/settings` - علامات إعدادات النظام (عامة، توجيه، إعدادات التحرير والإعدادات البرمجية، إلخ.) -- `/dashboard/api-manager` - دورة حياة مفتاح برمجة برمجة التطبيقات والأذونات النموذجية## سياق النظام عالي المستوى```mermaid - flowchart LR - subgraph Clients[Developer Clients] - C1[Claude Code] - C2[Codex CLI] - C3[OpenClaw / Droid / Cline / Continue / Roo] - C4[Custom OpenAI-compatible clients] - BROWSER[Browser Dashboard] - end +## Scope and Boundaries - subgraph Router[OmniRoute Local Process] - API[V1 Compatibility API\n/v1/*] - DASH[Dashboard + Management API\n/api/*] - CORE[SSE + Translation Core\nopen-sse + src/sse] - DB[(storage.sqlite)] - UDB[(usage tables + log artifacts)] - end +### In Scope - subgraph Upstreams[Upstream Providers] - P1[OAuth Providers\nClaude/Codex/Gemini/Qwen/Qoder/GitHub/Kiro/Cursor/Antigravity] - P2[API Key Providers\nOpenAI/Anthropic/OpenRouter/GLM/Kimi/MiniMax\nDeepSeek/Groq/xAI/Mistral/Perplexity\nTogether/Fireworks/Cerebras/Cohere/NVIDIA] - P3[Compatible Nodes\nOpenAI-compatible / Anthropic-compatible] - end +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration - subgraph Cloud[Optional Cloud Sync] - CLOUD[Cloud Sync Endpoint\nNEXT_PUBLIC_CLOUD_URL] - end +### Out of Scope - C1 --> API - C2 --> API - C3 --> API - C4 --> API - BROWSER --> DASH +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) - API --> CORE - DASH --> DB - CORE --> DB - CORE --> UDB +## Dashboard Surface (Current) - CORE --> P1 - CORE --> P2 - CORE --> P3 +Main pages under `src/app/(dashboard)/dashboard/`: - DASH --> CLOUD +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions -```` +## High-Level System Context + +```mermaid +flowchart LR + subgraph Clients[Developer Clients] + C1[Claude Code] + C2[Codex CLI] + C3[OpenClaw / Droid / Cline / Continue / Roo] + C4[Custom OpenAI-compatible clients] + BROWSER[Browser Dashboard] + end + + subgraph Router[OmniRoute Local Process] + API[V1 Compatibility API\n/v1/*] + DASH[Dashboard + Management API\n/api/*] + CORE[SSE + Translation Core\nopen-sse + src/sse] + DB[(storage.sqlite)] + UDB[(usage tables + log artifacts)] + end + + subgraph Upstreams[Upstream Providers] + P1[OAuth Providers\nClaude/Codex/Gemini/Qwen/Qoder/GitHub/Kiro/Cursor/Antigravity] + P2[API Key Providers\nOpenAI/Anthropic/OpenRouter/GLM/Kimi/MiniMax\nDeepSeek/Groq/xAI/Mistral/Perplexity\nTogether/Fireworks/Cerebras/Cohere/NVIDIA] + P3[Compatible Nodes\nOpenAI-compatible / Anthropic-compatible] + end + + subgraph Cloud[Optional Cloud Sync] + CLOUD[Cloud Sync Endpoint\nNEXT_PUBLIC_CLOUD_URL] + end + + C1 --> API + C2 --> API + C3 --> API + C4 --> API + BROWSER --> DASH + + API --> CORE + DASH --> DB + CORE --> DB + CORE --> UDB + + CORE --> P1 + CORE --> P2 + CORE --> P3 + + DASH --> CLOUD +``` ## Core Runtime Components ## 1) API and Routing Layer (Next.js App Routes) -الدلائل الرئيسية: +Main directories: -- `src/app/api/v1/*` و `src/app/api/v1beta/*` لواجهات برمجة التطبيقات المتوافقة -- `src/app/api/*` لواجهات برمجة تطبيقات للإدارة/التكوين -- إعادة الكتابة التالية في الخريطة `next.config.mjs` `/v1/*` إلى `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -طرق التوافق: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` - تشمل نماذج مخصصة ذات `مخصصة: صحيح` -- `src/app/api/v1/embeddings/route.ts` - إنشاء التضمين (6 مفري) -- `src/app/api/v1/images/ Generations/route.ts` - إنشاء الصور (4+ موفري خدمات بما في ذلك Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` - دردشة مخصصة لكل المرشحين -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` - عمليات التضمين المخصصة لكل المجالات -- `src/app/api/v1/providers/[provider]/images/ Generations/route.ts` - صور مخصصة لكل إطار +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` - `src/app/api/v1beta/models/[...path]/route.ts` -الفترات الإدارية: +Management domains: -- المصادقة/الإعدادات: `src/app/api/auth/*`، `src/app/api/settings/*` -- مقدمو الخدمة/الاتصالات: `src/app/api/providers*` -- عقد الموفر: `src/app/api/provider-nodes*` -- الروابط ذات الصلة: `src/app/api/provider-models` (GET/POST/DELETE) -- كتالوج الارتباطات: `src/app/api/models/route.ts` (GET) -- الوكيل التنفيذي: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` --لوحة المفاتيح/الأسماء المستعارة/المجموعات/التسعير: `src/app/api/keys*`، `src/app/api/models/alias`، `src/app/api/combos*`، `src/app/api/pricing` --استخدام: `src/app/api/usage/*` -- الناقلات/السحابة: `src/app/api/sync/*`، `src/app/api/cloud/*` -- مساعدي أدوات CLI: `src/app/api/cli-tools/*` -- مرشح IP: `src/app/api/settings/ip-filter` (GET/PUT) -- تكلفة التفكير: `src/app/api/settings/thinking-budget` (GET/PUT) -- متشوق النظام: `src/app/api/settings/system-prompt` (GET/PUT) -- الجلسات: `src/app/api/sessions` (GET) -- النطاق المعدل: `src/app/api/rate-limits` (GET) -- معطف: `src/app/api/resilience` (GET/PATCH) - ملفات تعريف الموفر، التفاضل والتكامل، حالة لا يمكن تعديلها -- إعادة ضبط ضبط: `src/app/api/resilience/reset` (POST) - إعادة ضبط القواطع + تخفيف التهدئة -- إحصائيات ذاكرة تخزين مؤقتة: `src/app/api/cache/stats` (GET/DELETE) -- توفر النموذج: `src/app/api/models/availability` (GET/POST) -- القياس عن بعد: `src/app/api/telemetry/summary` (GET) -- الميزانية: `src/app/api/usage/budget` (GET/POST) -- السلاسل الاحتياطية: `src/app/api/fallback/chains` (GET/POST/DELETE) -- تدقيق تماما: `src/app/api/compliance/audit-log` (GET) -- التقييمات: `src/app/api/evals` (GET/POST)، `src/app/api/evals/[suiteId]` (GET) -- للمزيد: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -وحدات السرعة الرئيسية:- الإدخال: `src/sse/handlers/chat.ts` -- أريد الأساسي: `open-sse/handlers/chatCore.ts` -- محولات تنفيذ الموفر: `open-sse/executors/*` -- الاكتشاف الجديد/تكوين الموفر: `open-sse/services/provider.ts` -- تحليل/حل النموذج: `src/sse/services/model.ts`، `open-sse/services/model.ts` -- الحساب الاحتياطي للحساب: `open-sse/services/accountFallback.ts` -- سجل الترجمة: `open-sse/translator/index.ts` -- تحويلات الدفق: `open-sse/utils/stream.ts`، `open-sse/utils/streamHandler.ts` --الطلب/تطبيع الاستخدام: `open-sse/utils/usageTracking.ts` -- فكر في محلل العناوين: `open-sse/utils/thinkTagParser.ts` --معالج التضمين: `open-sse/handlers/embeddings.ts` -- سجل موفر التضمين: open-sse/config/embeddingRegistry.ts --معالج إنشاء الصور: `open-sse/handlers/imageGeneration.ts` -- سجل موفر الصور: `open-sse/config/imageRegistry.ts` -- تعريف القيمة: `open-sse/handlers/responseSanitizer.ts` -- تطبيع الدور: `open-sse/services/roleNormalizer.ts` +## 2) SSE + Translation Core -الخدمات (منطقة الأعمال): +Main flow modules: -- اختيار الحساب/تسجيل النقاط: `open-sse/services/accountSelector.ts` -- إدارة دورة حياة السياق: `open-sse/services/contextManager.ts` -- فرض مرشح IP: `open-sse/services/ipFilter.ts` -- تعقيب النظر: `open-sse/services/sessionManager.ts` --طلب إلغاء البيانات المكررة: `open-sse/services/signatureCache.ts` -- البناء الكامل: `open-sse/services/systemPrompt.ts` -- التفكير في إدارة الميزانية: `open-sse/services/thinkingBudget.ts` -- توجيه نموذج حرف البدل: `open-sse/services/wildcardRouter.ts` -- إدارة إلى حد التعديل: `open-sse/services/rateLimitManager.ts` -- قاطع الدائرة: `open-sse/services/circuitBreaker.ts` +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -وحدات المجال: +Services (business logic): -- توفر النموذج: `src/lib/domain/modelAvailability.ts` -- متطلبات/ميزانيات التكلفة: `src/lib/domain/costRules.ts` -- السياسة الافتراضية: `src/lib/domain/fallbackPolicy.ts` -- محلل التحرير والسرد: `src/lib/domain/comboResolver.ts` -- تأمين التأمين: `src/lib/domain/lockoutPolicy.ts` -- محرك السياسة: `src/domain/policyEngine.ts` - القفل المركزي ← الميزانية ← التقييم الاحتياطي -- كتالوج الرموز لسبب: `src/lib/domain/errorCodes.ts` -- معرف الطلب: `src/lib/domain/requestId.ts` -- مهلة الجلب: `src/lib/domain/fetchTimeout.ts` --طلب القياس عن بعد: `src/lib/domain/requestTelemetry.ts` -- شامل/الدقيق: `src/lib/domain/compliance/index.ts` -- عداء التقييم: `src/lib/domain/evalRunner.ts` -- دونية المجال المجال: `src/lib/db/domainState.ts` - SQLite CRUD للسلاسل الاحتياطية، والميزانيات، خسر التكلفة، وحالة القفل، وقواطع الضوء +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -وحدات موفر OAuth (12 ملفًا فرديًا ضمن `src/lib/oauth/providers/`): +Domain layer modules: -- فهرس التسجيل: `src/lib/oauth/providers/index.ts` -- مقدمو الخدمات الأشخاص: `claude.ts`، `codex.ts`، `gemini.ts`، `antigravity.ts`، `qode.ts`، `qwen.ts`، `kimi-coding.ts`، `github.ts`، `kiro.ts`، `cursor.ts`، `kilocode.ts`، `cline.ts` -- طعام السباحة: `src/lib/oauth/providers.ts` - يُعاد تصديره من العناصر العناصر## 3) طبقة الثبات +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -قاعدة بيانات الحالة الأساسية (SQLite):- المعرفة البشرية الأساسية: `src/lib/db/core.ts` (better-sqlite3، migrations، WAL) -- واجهة إعادة التصدير: `src/lib/localDb.ts` (طبقة توافق مختلفة للمتصلين) -- الملف: `${DATA_DIR}/storage.sqlite` (أو `$XDG_CONFIG_HOME/omniroute/storage.sqlite` عند الضرورة، وإلا `~/.omniroute/storage.sqlite`) -- كيانات (الجداول + أسماء KV): ProvideConnections، وproviderNodes، وmodelAliases، والمجموعات، WapiKeys، والإعدادات، والتسعير،**customModels**،**proxyConfig**،**ipFilter**،**thinkingBudget**،**systemPrompt** +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -بمرور الوقت الاستخدام: +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -- الواجهة: `src/lib/usageDb.ts` (وحدات متحللة في `src/lib/usage/*`) -- جداول SQLite في `storage.sqlite`: `usage_history`، `call_logs`، `proxy_logs` -- تبرز عناصر الملف الاختياري للتوافق/تصحيح سبب (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- يتم رحيل ملفات JSON القديمة إلى SQLite عن طريق عمليات رحيل بدء التشغيل عند وجودها +## 3) Persistence Layer -قاعدة بيانات المجال (SQLite): +Primary state DB (SQLite): -- `src/lib/db/domainState.ts` - عمليات إنتاج CRUD لحالة المجال -- الجداول (التي تم تحديدها في `src/lib/db/core.ts`): `domain_fallback_chains`، `domain_budgets`، `domain_cost_history`، `domain_lockout_state`، `domain_circuit_breakers`. -- نمط ذاكرة التخزين المؤقت للكتابة: قرص الاتصال موجود في الذاكرة الموثوقة في وقت التشغيل؛ تتم كتابة الطفرات بشكل متزامن إلى SQLite؛ يتم استعادة حالة قاعدة البيانات عند البداية الباردة ## 4) المصادقة + الأسطح الأمنية +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -- مصادقة ملف تعريف الارتباط في لوحة المعلومات: `src/proxy.ts`، `src/app/api/auth/login/route.ts` -- إنشاء/التحقق من مفتاح واجهة برمجة التطبيقات: `src/shared/utils/apiKey.ts` --أسرار الموفر في الخطوط "providerConnections". -- دعم خارجي تمامًا عبر `open-sse/utils/proxyFetch.ts` (env vars) و`open-sse/utils/networkProxy.ts` (قابل للتكوين لكل المرشحين أو عالمي)## 5) Cloud Sync +Usage persistence: -- جدولة init: `src/lib/initCloudSync.ts`، `src/shared/services/initializeCloudSync.ts`، `src/shared/services/modelSyncScheduler.ts` -- أهم الأحداث: `src/shared/services/cloudSyncScheduler.ts` -- أهم الأحداث: `src/shared/services/modelSyncScheduler.ts` -- التحكم في المسار: `src/app/api/sync/cloud/route.ts`## دورة حياة الطلب (`/v1/chat/completions`)```mermaid +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present + +Domain State DB (SQLite): + +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) + +```mermaid sequenceDiagram autonumber participant Client as CLI/SDK Client @@ -297,7 +331,7 @@ sequenceDiagram Stream-->>Client: SSE chunks / JSON response Stream->>Usage: extract usage + persist history/log -```` +``` ## Combo + Account Fallback Flow @@ -329,15 +363,19 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -يتم اتخاذ القرار الاحتياطي بواسطة `open-sse/services/accountFallback.ts` باستخدام رموز الحالة للاستدلال على رسائل الخطأ. تسهيل توجيه التشغيل والتنسيق بين الطرفين طوعًا للمساعدة في تقديم الطلبات: يتم التعامل مع 400s على نطاق الموفر مثل كتلة المحتوى الأول وفشل التحقق من صحة الدور على أنها فشل رئيسي للنموذج، لذا لا يزال لا يزال مطلوبًا التحرير والسرد التالي.## OAuth Onboarding and Token Refresh Lifecycle```mermaid +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle + +```mermaid sequenceDiagram -autonumber -participant UI as Dashboard UI -participant OAuth as /api/oauth/[provider]/[action] -participant ProvAuth as Provider Auth Server -participant DB as localDb -participant Test as /api/providers/[id]/test -participant Exec as Provider Executor + autonumber + participant UI as Dashboard UI + participant OAuth as /api/oauth/[provider]/[action] + participant ProvAuth as Provider Auth Server + participant DB as localDb + participant Test as /api/providers/[id]/test + participant Exec as Provider Executor UI->>OAuth: GET authorize or device-code OAuth->>ProvAuth: create auth/device flow @@ -355,10 +393,13 @@ participant Exec as Provider Executor Exec-->>Test: valid or refreshed token info Test->>DB: update status/tokens/errors Test-->>UI: validation result +``` -```` +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. -يتم تنفيذ التحديث أثناء حركة التحرير المباشر داخل `open-sse/handlers/chatCore.ts` عبر المنفذ `refreshCredentials()`.## دورة حياة المزامنة السحابية (تمكين / مزامنة / تعطيل)```mermaid +## Cloud Sync Lifecycle (Enable / Sync / Disable) + +```mermaid sequenceDiagram autonumber participant UI as Endpoint Page UI @@ -386,13 +427,17 @@ sequenceDiagram Sync->>Cloud: DELETE /sync/{machineId} Sync->>Claude: switch ANTHROPIC_BASE_URL back to local (if needed) Sync-->>UI: disabled -```` +``` -يتم تشغيل الدورية بواسطة "CloudSyncScheduler" عند السحابة.## نموذج البيانات وخريطة التخزين```mermaid +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map + +```mermaid erDiagram -SETTINGS ||--o{ PROVIDER_CONNECTION : controls -PROVIDER_NODE ||--o{ PROVIDER_CONNECTION : backs_compatible_provider -PROVIDER_CONNECTION ||--o{ USAGE_ENTRY : emits_usage + SETTINGS ||--o{ PROVIDER_CONNECTION : controls + PROVIDER_NODE ||--o{ PROVIDER_CONNECTION : backs_compatible_provider + PROVIDER_CONNECTION ||--o{ USAGE_ENTRY : emits_usage SETTINGS { boolean cloudEnabled @@ -485,15 +530,18 @@ PROVIDER_CONNECTION ||--o{ USAGE_ENTRY : emits_usage string prompt string position } +``` -```` +Physical storage files: -ملفات الوضع المالي: +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` -- قاعدة بيانات وقت التشغيل الأساسي: `${DATA_DIR}/storage.sqlite` -- أسطر سجل الطلب: `${DATA_DIR}/log.txt` (أداة متوافقة/تصحيح سبب) -- أرشيفات استضافة المؤتمرات التنظيمية: `${DATA_DIR}/call_logs/` -- مجموعات تصحيح الأخطاء المترجم/الطلب الاختيارية: `/logs/...`## Deployment Topology```mermaid +## Deployment Topology + +```mermaid flowchart LR subgraph LocalHost[Developer Host] CLI[CLI Tools] @@ -520,200 +568,255 @@ flowchart LR Core --> UsageDB Core --> Providers Next --> SyncCloud -```` +``` ## Module Mapping (Decision-Critical) ### Route and API Modules -- `src/app/api/v1/*`، `src/app/api/v1beta/*`: واجهات برمجة التطبيقات المتوافقة -- `src/app/api/v1/providers/[provider]/*`: مسارات مخصصة لكل دليل (الدردشة والتضمينات والصور) -- `src/app/api/providers*`: موفر CRUD، التحقق من الصحة، الاختبار -- `src/app/api/provider-nodes*`: إدارة العقد المتوافقة المخصصة -- `src/app/api/provider-models`: إدارة الارتباطات المخصصة (CRUD) -- `src/app/api/models/route.ts`: برمجة تطبيقات كتالوج الارتباطات (الأسماء المستعارة + الارتباطات البديلة) -- `src/app/api/oauth/*`: تدفقات رمز OAuth/الجهاز -- `src/app/api/keys*`: دورة حياة مفتاح برمجة التطبيقات المحلية -- `src/app/api/models/alias`: إدارة الأسماء المستعارة -- `src/app/api/combos*`: إدارة التحرير والسرد الاحتياطي -- `src/app/api/pricing`: تجاوزات التسعير لحساب التكلفة -- `src/app/api/settings/proxy`: الصارم المعتمد (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: اختبار تشغيل الوكيل (POST) -- `src/app/api/usage/*`: واجهات برمجة تطبيقات الاستخدام والسجلات -- `src/app/api/sync/*` + `src/app/api/cloud/*`: نوبات السحابية والمساعدون الذين يتحملون السحابة -- `src/app/api/cli-tools/*`: كاتب/أداة الدما لتكوين CLI المحلي -- `src/app/api/settings/ip-filter`: قائمة IP مخصصة لها/القائمة المبتكرة (GET/PUT) -- `src/app/api/settings/thinking-budget`: الاختيار المناسب رمز التفكير (GET/PUT) -- `src/app/api/settings/system-prompt`: موجه النظام العام (GET/PUT) -- `src/app/api/sessions`: قائمة العناصر العضوية (GET) -- `src/app/api/rate-limits`: حالة لا يمكن تعديلها لكل حساب (GET)### التوجيه والتنفيذ الأساسي +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: تحليل الطلب، ومعالجة التحرير والسرد، حلقة الحساب -- `open-sse/handlers/chatCore.ts`: الترجمة، المنفذ، إعادة المحاولة/التحديث، إعداد الدفق -- `open-sse/executors/*`: التحكم الشبكة والتنسيق الخاص بالموفر### سجل الترجمة ومحولات التنسيق +### Routing and Execution Core -- `open-sse/translator/index.ts`: تسجيل المترجم وتنسيقه - -طلب المترجمين: `open-sse/translator/request/*` -- مترجمو المصدر: `open-sse/translator/response/*` -- ثوابت عادة: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: تفعيل/الحالة الفعالة واستمرارية المجال على SQLite -- `src/lib/localDb.ts`: إعادة تصدير التوافق لوحدات قاعدة البيانات -- `src/lib/usageDb.ts`: واجهة سجل/سجلات استخدامات المكالمات أعلى جداول SQLite## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -| يحتوي على كل موفر على منفذ تنفيذي متخصص لعدة `BaseExecutor` (في `open-sse/executors/base.ts`)، والذي يوفر بيانات إنشاء عنوان URL، ولكنه، جاهز المحاولة مع الأسيي، ومآثر تحديث الاعتماد، وطريقة استمرار `execute()`. | المنفذ | المزود (المقدمون) | التعامل الخاص | -| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------- | ------------- | -| `المنفذ الافتراضي` | أوبن إيه آي، كلود، جيميني، كوين، كيودر، أوبن روتر، جي إل إم، كيمي، ميني ماكس، ديب سيك، جروك، إكس آي آي، ميسترال، بيربليكسيتي، توغا، فاير ووركس، سيريبراس، كوهير، نفيديا | الاختيارية عنوان URL/الرأس الكيميائي لكل | -| `منفذ مضاد للجغرافيا` | جوجل مكافحة الجاذبية | معرفات المشروع/الجلسة المخصصة، إعادة المحاولة بعد التحليل | -| `منفذ الكودكس` | OpenAI Codex | يحقن تعليمات النظام، ويفرض جهدًا منطقيًا | -| `منفذ مفصل` | بيئة تطوير متكاملة للمؤشر | البروتوكول ConnectRPC، ترجمة Protobuf، طلب التوقيع عبر الفصول الاختباري | -| `GithubExecutor` | جيثب مساعد الطيار | تحديث الرمز المميز لـ Copilot، ورؤوس محاكاة VSCode | -| `KiroExecutor` | AWS CodeWhisperer/كيرو | يتغير الثنائي لـ AWS EventStream → تحويل SSE | -| `الجوزاءCLIEExecutor` | الجوزاء CLI | دورة تحديث رمز OAuth المميز لـ Google | +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| يستخدم جميع الموفرين الآخرين (بما في ذلك العقد المتوافق المخصص) "DefaultExecutor".## مصفوفة توافق الموفرين | مقدم | التنسيق | مصادقة | تيار مستمر | غير دفق | تحديث الرمز المميز | برمجة تطبيقات الاستخدام | -| ---------------------------------------------------------------------------------------------------------- | ---------------- | -------------------------------------- | --------------- | ---------- | ------- | ----------------------- | ----------------------- | -| كلود | كلود | واجهة برمجة التطبيقات الرئيسية / OAuth | ✅ | ✅ | ✅ | ⚠️ المشرف فقط | -| الجوزاء | الجوزاء | واجهة برمجة التطبيقات الرئيسية / OAuth | ✅ | ✅ | ✅ | ⚠️ وحدة التحكم السحابية | -| الجوزاء CLI | الجوزاء-cli | أووث | ✅ | ✅ | ✅ | ⚠️ وحدة التحكم السحابية | -| مكافحة الجاذبية | ضد الجاذبية | أووث | ✅ | ✅ | ✅ | ✅ الحصة الكاملة API | -| أوبن آي | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ | -| الدستور الغذائي | openai-responses | أووث | ✅ مجبور | ❌ | ✅ | ✅الحدود المعدلة | -| جيثب مساعد الطيار | أوبيناي | OAuth + رمز مساعد الطيار | ✅ | ✅ | ✅ | ✅ لقطات الحصص | -| | مؤثر | مؤثر الاستطلاع المفضل | ✅ | ✅ | ❌ | ❌ | -| كيرو | كيرو | AWS SSO OIDC | ✅(ايفنت ستريم) | ❌ | ✅ | ✅ حدود الاستخدام | -| كوين | أوبيناي | أووث | ✅ | ✅ | ✅ | ⚠️ طلب حسب الطلب | -| قدير | أوبيناي | OAuth (أساسي) | ✅ | ✅ | ✅ | ⚠️ طلب حسب الطلب | -| اوبن راوتر | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ | -| جي إل إم/كيمي/ميني ماكس | كلود | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ | -| ديب سيك | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ | -| جروك | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ | -| xAI (جروك) | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ | -| ميسترال | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ | -| الحيرة | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ | -| منظمة العفو الدولية | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ | -| منظمة العفو الدولية للعبة | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ | -| الشيخ | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ | -| كوهير | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ | -| نفيديا نيم | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ | ## تنسيق تغطية الترجمة | +### Persistence -تتضمن التنسيقات المصدر المكتشفة ما يلي: +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -- `أوبيني` -- `الردود المفتوحة` -- "كلود". -- "الجوزاء". +## Provider Executor Coverage (Strategy Pattern) -تتضمن الواردات التفصيلية ما يلي: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. -- دردشة/ردود OpenAI -- كلود - -الجوزاء/الجوزاء-CLI/الظرف للجاذبية -- كيرو -- مرض +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | -استخدم الترجمات**OpenAI كتنسيق مركزي**— جرب جميع التحويلات عبر OpenAI كتنسيق وسيط:` -تنسيق المصدر → OpenAI (المحور) → التنسيق المستهدف` +All other providers (including custom compatible nodes) use the `DefaultExecutor`. -يتم تحديد الترجمات ديناميكيًا استنادًا إلى شكل حمولة المصدر والتنسيق المستهدف للموفر. +## Provider Compatibility Matrix -طبقات معالجة إضافية في مسار الترجمة: +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | --**تطهير الاستجابة**— يزيل الحقول غير القياسية من استجابات تنسيق OpenAI (سواء المتدفقة أو غير المتدفقة) لضمان الامتثال الصارم لـ SDK -**تطبيع الدور**— تحويل `المطور` ← `النظام` للأهداف غير التابعة لـ OpenAI؛ يدمج "النظام" → "المستخدم" للنماذج التي ترفض دور النظام (GLM، ERNIE) -**استخراج علامة التفكير**— يوزع كتل `...` من المحتوى إلى حقل `reasoning_content` -**الإخراج المنظم**— يحول OpenAI `response_format.json_schema` إلى `responseMimeType` + `responseSchema` الخاص بـ Gemini## Supported API Endpoints +## Format Translation Coverage -| نقطة النهاية | تنسيق | معالج | -| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------------- | ----------------- | -| `POST /v1/chat/completions` | دردشة OpenAI | `src/sse/handlers/chat.ts` | -| `POST /v1/messages` | رسائل كلود | نفس المعالج (تم اكتشافه تلقائيًا) | -| `POST /v1/responses` | ردود OpenAI | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/embeddings` | تضمينات OpenAI | `open-sse/handlers/embeddings.ts` | -| `الحصول على /v1/embeddings` | قائمة النماذج | طريق API | -| `POST /v1/images/أجيال` | صور OpenAI | `open-sse/handlers/imageGeneration.ts` | -| `الحصول على /v1/images/أجيال` | قائمة النماذج | طريق API | -| `POST /v1/providers/{provider}/chat/completions` | دردشة OpenAI | مخصص لكل مزود مع التحقق من صحة النموذج | -| `POST /v1/providers/{provider}/embeddings` | تضمينات OpenAI | مخصص لكل مزود مع التحقق من صحة النموذج | -| `POST /v1/providers/{provider}/images/generations` | صور OpenAI | مخصص لكل مزود مع التحقق من صحة النموذج | -| `POST /v1/messages/count_tokens` | عدد كلود توكن | طريق API | -| `الحصول على /v1/models` | قائمة نماذج OpenAI | مسار واجهة برمجة التطبيقات (الدردشة + التضمين + الصورة + النماذج المخصصة) | -| `الحصول على /api/models/catalog` | كتالوج | جميع النماذج مجمعة حسب الموفر + النوع | -| `POST /v1beta/models/*:streamGenerateContent` | مولود برج الجوزاء | طريق API | -| `الحصول على/PUT/DELETE /api/settings/proxy` | تكوين الوكيل | تكوين وكيل الشبكة | -| `POST /api/settings/proxy/test` | اتصال الوكيل | نقطة نهاية اختبار صحة الوكيل/الاتصال | -| `الحصول على/النشر/الحذف /api/provider-models` | نماذج المزود | البيانات الوصفية لنموذج الموفر تدعم النماذج المتاحة المخصصة والمدارة | ## Bypass Handler | +Detected source formats include: -يعترض معالج التجاوز (`open-sse/utils/bypassHandler.ts`) طلبات "رمية سريعة" معروفة من Claude CLI - أصوات التمهيد، واستخراج العناوين، وعدد الرموز المميزة - ويعيد**استجابة زائفة**دون استهلاك الرموز المميزة للموفر الرئيسي. يتم تشغيل هذا فقط عندما يحتوي "User-Agent" على "clude-cli".## Request Logger Pipeline +- `openai` +- `openai-responses` +- `claude` +- `gemini` -يوفر مسجل الطلب (`open-sse/utils/requestLogger.ts`) مسارًا لتسجيل تصحيح الأخطاء مكون من 7 مراحل، معطل افتراضيًا، وممكن عبر `ENABLE_REQUEST_LOGS=true`:``` +Target formats include: + +- OpenAI chat/Responses +- Claude +- Gemini/Gemini-CLI/Antigravity envelope +- Kiro +- Cursor + +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` +Source Format → OpenAI (hub) → Target Format +``` + +Translations are selected dynamically based on source payload shape and provider target format. + +Additional processing layers in the translation pipeline: + +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` + +## Supported API Endpoints + +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | + +## Bypass Handler + +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt - ``` -تتم كتابة الملفات إلى `/logs//` لكل جلسة طلب.## أوضاع الفشل والمرونة## 1) Account/Provider Availability +Files are written to `/logs//` for each request session. -- عبارة عن حساب الموفر عند أخطاء/معدل/مصادقة -- إرجاع الحساب قبل فشل الطلب -- نموذج التحرير والسرد الاحتياطي عند استنفاد مسار النموذج/المزود الحالي## 2) Token Expiry +## Failure Modes and Resilience -- ملفات التقدم والتحديث مع إعادة محاولة توفير خدمة موثوقة للتحديث -- 401/403 إعادة المحاولة بعد محاولة التحديث في المسار الأساسي## 3) Stream Safety +## 1) Account/Provider Availability -- وحدة تحكم قطع الاتصال بالتيار المستمر -- دفق الترجمة تدفق مع نهاية الدفق و `[تم]` -- ترخيص للاستخدام عندما تكون البيانات الوصفية للاستخدام الموفر المفقود## 4) تدهور المزامنة السحابية +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- أخطاء الأخطاء ولكن استمر تشغيلها محليًا -- يحتوي على المجدول على منطقه قادر على إعادة المحاولة، ولكن التنفيذ الدوري يستدعي حاليا متزامنة التفعيل بشكل افتراضي## 5) Data Integrity +## 2) Token Expiry -- عمليات ترحيل مخطط SQLite وفواتير الترقية التلقائية عند بدء التشغيل -- JSON القديم → مسار التوافق ترحيل SQLite## إمكانية المراقبة والإشارات التشغيلية +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -مصادر معرفة وقت التشغيل: +## 3) Stream Safety -- أرشيف وحدة التحكم من `src/sse/utils/logger.ts` -- مجاميع الاستخدام لكل طلب في SQLite (`usage_history`، `call_logs`، `proxy_logs`) -- التقاط التفاصيل الصافية الصافية على أربع مراحل في SQLite (`request_detail_logs`) عندما تكون `settings.detailed_logs_enabled=true` -- سجل حالة الطلب النصي في "log.txt" (اختياري/متوافق) -- سجلات الطلب/الترجمة المتخصصة الاختيارية ضمن `السجلات/` عندما يكون `ENABLE_REQUEST_LOGS=true` -- نقاط نهاية استخدام معلومات اللوحة (`/api/usage/*`) لاستهلاك واجهة المستخدم +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -يقوم بالتقاط تكتيكات متعددة بتخزين ما يصل إلى أربع مراحل من نشاطات JSON لكل ما يستقبل بصرية: +## 4) Cloud Sync Degradation -- الطلب الوارد من العميل -- تم إرسال الطلب المترجم إلى المنبع -- إعادة بناء الرابط الموفر JSON؛ يتم ضغط الاستجابات المتدفقة إلى الملخص النهائي بالإضافة إلى بيانات تعريف الدفق --الرد النهائي الذي تم إرجاعه بواسطة OmniRoute؛ يتم تخزين الاستجابات المتدفقة في نفس النموذج الملخص المكون## الحدود الحساسة للأمان +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -- يعمل سر JWT (`JWT_SECRET`) على تأمين المصادقة/التوقيع على ملف تعريف الارتباط لجلسة لوحة المعلومات -- يجب الالتزام بالبراءة الأولية لكلمة المرور (`INITIAL_PASSWORD`) ووافق على الاعتراف بها لأول مرة -- يعمل سر HMAC لمفتاح API (`API_KEY_SECRET`) على تنسيق تنسيق مفتاح API المحلي الذي تم التعاقد معه -- تظلل أسرار الموفر (مفاتيح/رموز برمجة التطبيقات) موجودة في قاعدة البيانات الأصلية وحماتها على مستوى نظام الملفات -- تعتمد نقاط نهاية الهجمات السحابية على مصادقة مفتاح API + دلالات معرف الجهاز## مصفوفة البيئة ووقت التشغيل +## 5) Data Integrity -تحريرات البيئة المستخدمة بشكل نشط بواسطة تعليمات الحظر:- التطبيق/المصادقة: `JWT_SECRET`، `INITIAL_PASSWORD` -- التخزين: `DATA_DIR` -- العقدة المتوافقة: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- تجاوز قاعدة الاختيار الاختيارية (Linux/macOS عند إلغاء تعيين `DATA_DIR`): `XDG_CONFIG_HOME` -- التجزئة الأمنية: `API_KEY_SECRET`، `MACHINE_ID_SALT` -- التسجيل: `ENABLE_REQUEST_LOGS` -- عناوين URL للاستقبال/السحابة: `NEXT_PUBLIC_BASE_URL`، `NEXT_PUBLIC_CLOUD_URL` -- الوكيل الشامل: `HTTP_PROXY`، `HTTPS_PROXY`، `ALL_PROXY`، `NO_PROXY` ومتغيرات الصغيرة الصغيرة -- علامات ميزات SOCKS5: `ENABLE_SOCKS5_PROXY`، `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- مساعدو النظام الأساسي/وقت التشغيل (وليس تفعيل الخاص بالتطبيق): `APPDATA`، `NODE_ENV`، `PORT`، `HOSTNAME`## الملاحظات المعمارية المعروفة +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -1. تشارك `usageDb` و`localDb` في نفس الدليل الأساسي (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> ``~/.omniroute`) مع ترحيل الملفات القديمة. -2. يفوض `/api/v1/route.ts` إلى نفس منشئ الكتالوج الموحد الذي يستخدمه `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) العلم الانحراف الدلالي. -3. يقوم بطلب تسجيل بكتابة الرؤوس/النص الكامل عند جاكسونه؛ التعامل مع سجل الدليل على أنه حساسية. -4. يعتمد حماية السحابة على `NEXT_PUBLIC_BASE_URL` صحيح وإمكانية الوصول إلى نقطة نهاية السحابة. -5. تم نشر الدليل `open-sse/` باسم `@omniroute/open-sse`**حزمة مساحة العمل npm**. يقوم بكود المصدر باستيراده عبر `@omniroute/open-sse/...` (تم حله بواسطة Next.js `transpilePackages`). لا تسلك الطرق المستمرة في هذا المستند استخدم اسم الدليل `open-sse/` للاتساق. -6. نستخدم الكائنات الموجودة في لوحة المعلومات**Recharts**(المستندة إلى SVG) لتصورات التحليلات التفاعلية التي يمكن الوصول إليها (المخططات الشريطية للاستخدام للنموذج، والجرافيك المستخدمة للمخرجين مع النجاح). -7.استخدام السيولة E2E**Playwright**(`tests/e2e/`)، ويمكنها عبر `npm run test:e2e`. المستخدمة في الوحدة**Node.js test runner**(`tests/unit/`)، ويمكن تشغيلها عبر `npm run test:unit`. كود المصدر ضمن `src/` هو**TypeScript**(`.ts`/`.tsx`)؛ تختلف مساحة العمل `open-sse/` JavaScript (`.js`). -8. تم ضبط صفحة الإعدادات في 5 علامات: الأمان، التوجيه (6 إستراتيجيات عالمية: التعبئة العامة، جولة روبن، p2c، تنظيم غير محدد لاستخدامًا، تحسين التكلفة)، اشتراك (حدود الرسوم المتحركة للتحرير، قطع الدقة، إبداع)، الذكاء الاصطناعي (ميزانية التفكير، متشوق للنظام، ذاكرة التخزين المؤقت السريع)، المتقدمة (الوكيل).## قائمة التحقق من التشغيل +## Observability and Operational Signals -- البناء من المصدر: ``npm run build`` -- إنشاء صورة Docker: `docker build -t omniroute .` -- بدء الخدمة والتحقق: -- `الحصول على /api/settings` -- `الحصول على /api/v1/models` -- يجب أن يكون عنوان URL الأساسي لهدف واجهة سطر اللاسلكي هو `http://:20128/v1` عندما يكون `PORT=20128` -``` +Runtime visibility sources: + +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption + +Detailed request payload capture stores up to four JSON payload stages per routed call: + +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` +- `GET /api/v1/models` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/ar/docs/FEATURES.md b/docs/i18n/ar/docs/FEATURES.md index b975bc2f06..0db3f5758e 100644 --- a/docs/i18n/ar/docs/FEATURES.md +++ b/docs/i18n/ar/docs/FEATURES.md @@ -4,70 +4,168 @@ --- -دليل مرئي لكل قسم من معلومات لوحة OmniRoute.---## 🔌 Providers -إدارة اتصالات الذكاء الصناعي: موفري OAuth (Claude Code وCodex وGemini CLI) وموفري مفاتيح API (Groq وDeepSeek وOpenRouter) ومقدمي خدمات العيد (Qoder وQwen وKiro). لحسابات كيرو على تتبع الاعتماد الائتماني - الأرصدة النهائية لإجمالي استطلاعات الرأي المتخصصة في لوحة التحكم → استخدام.![Providers Dashboard](screenshots/01-providers.png)--- + +Visual guide to every section of the OmniRoute dashboard. + +--- + +## 🔌 Providers + +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) + +--- ## 🎨 Combos -أنشئ مجموعات التوجيه باستخدام 6 إستراتيجيات: نأمل، والمتزايدة، والدورية، والعشوائية، وأقل استخدامًا، والمُحسّن من حيث التكلفة. وخاصة مجموعة نماذج متعددة مع اختلافات سريعة وفحوصات للجاهزية.![Combos Dashboard](screenshots/02-combos.png)--- +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) + +--- ## 📊 Analytics -تحليلات استخدام شاملة مع الرمز المميز، وتقديرات التكلفة، وخرائط، ومخططات التوزيع الأسبوعية، والتفاصيل لكل محمية.![Analytics Dashboard](screenshots/03-analytics.png)--- +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) + +--- ## 🏥 System Health -التسجيل في الوقت الفعلي: وقت العمل، والذاكرة، والإصدار، والنسب لزمن الوصول (p50/p95/p99)، وإحصائيات ذاكرة التخزين المؤقتة، وحالات منع دائرة الموفر.![Health Dashboard](screenshots/04-health.png)--- +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) + +--- ## 🔧 Translator Playground -أدوات لتصحيح أخطاء ترجمات برمجة التطبيقات:**ساحة اللعب**(محول أربعة نجاح)،**اختبار الدردشة**(الطلب المباشر)،**منصة الاختبار**(اختبارات الدفعة)، و**المراقب المباشر**(بث الوقت في العمل).![Translator Playground](screenshots/05-translator.png)--- +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) + +--- ## 🎮 Model Playground _(v2.0.9+)_ -اختبر أي نموذج مباشرة من لوحة القيادة. حدد الموفر والطراز والنقطة النهائية، وكتب المطالبات باستخدام محرر موناكو، وقم بتفعيل الاستثناءات في المنتج الفعلي، وإلغاء منتصف الدفق، والمعايرة التقليدية مرة.---## 🎨 Themes _(v2.0.5+)_ +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. -ألوان قابلة للتخصيص لمعلومات لوحة المفاتيح بأكملها. اختر من بين 7 ألوان محددة ليمين (مرجاني، أزرق، أخضر، بنفسجي، لون أحمر، سماوي) أو قم باختيار سمة مخصصة عن طريق اختيار أي سداسي عشري. يدعم وضع الضوء والظلام النظام.---## ⚙️ Settings +--- -لوحة الإعدادات شاملة مع علامات التبويب: +## 🎨 Themes _(v2.0.5+)_ --**عام**— تخزين النظام، وإدارة النسخ الاحتياطي (قاعدة بيانات التصدير/الاستيراد) -**المظهر**— محدد السماعة (داكن/فاتح/نظام)، الإعدادات المسبقة لموضوع الألوان والألوان المخصصة، ورؤية السجل الصحي، وعناصر التحكم في رؤية عنصر الشريط الجانبي -**الأمان**— حماية نقطة نهاية واجهة برمجة التطبيقات، وحظر الموفر المخصص، وتصفية IP، ومعلومات الاتصال -**التوجيه**— الأسماء المستعارة للنماذج، و الابتكارات الخلفية -**المرونة**— ونتيجة لذلك الحد الأقصى للمعدل، وضبط القيود، والتعطيل التلقائي للحسابات المحظورة، وانتهاء صلاحية الموفر -**متقدم**— تجاوز، ومسار تدقيق فقط، وتطبيق التدمير الاحتياطي![Settings Dashboard](screenshots/06-settings.png)--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- + +## ⚙️ Settings + +Comprehensive settings panel with tabs: + +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) + +--- ## 🔧 CLI Tools -ختمة واحدة لأدوات تميز الذكاء الصناعي: Claude Code، وCodex CLI، وGemini CLI، وOpenClaw، وKilo Code، وAntigravity، وCline، وContinue، وCursor، وFactory Droid. تم تفعيل/إعادة ضبط تلقائي، فقط تعريف الاتصال، والنتائج المباشرة.![CLI Tools Dashboard](screenshots/07-cli-tools.png)--- +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) + +--- ## 🤖 CLI Agents _(v2.0.11+)_ -لوحة معلومات للتحكم في وكلاء CLI. تم عرض شبكة مكونة من 14 وكيلًا مدمجًا (Codex وClaude وGoose وGemini CLI وOpenClaw وAider وOpenCode وCline وQwen Code وForgeCode وAmazon Q وOpen Interpreter وCursor CLI وWarp) مع: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**حالة التثبيت**— تم التثبيت/لم يتم العثور عليه باستخدام اكتشاف الإصدار -**توصيات المذكورة**— stdio، HTTP، وما إلى ذلك. -**الوكلاء يستهدفون**— هل هناك أي أداة لواجهة سطر الوكيل (CLI) عبر النموذج (الاسم، ثنائي، أمر الإصدار، وسيط النشر) -**مطابقة بصمة CLI**— التبديل لكل المرشحين لمطابقة توقيعات طلب CLI الأصلية، مما سيقدر من المبدع بالفعل مع ضمان عنوان IP الوكيل---## 🖼️ Media _(v2.0.3+)_ +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP -موجود في الصور ومقاطع الفيديو والموسيقى من لوحة التحكم. يدعم OpenAI وxAI وTogether وHyperbolic وSD WebUI وComfyUI وAnimateDiff وStable Audio Open وMusicGen.---## 📝 Request Logs +--- -تسجيل طلبات الإنتاج في الواقع باستخدام التصفية حسب الموفر والطراز والحساب ومفتاح واجهة برمجة التطبيقات. معلمات القيمة الناتجة عن التعويض الطبيعي ووقت التعويض وتفاصيل التعويض.![Usage Logs](screenshots/08-usage.png)--- +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- + +## 🖼️ Media _(v2.0.3+)_ + +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- + +## 📝 Request Logs + +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) + +--- ## 🌐 API Endpoint -نقطة نهاية واجهة برمجة التطبيقات الموحدة الخاصة بك مع تفاصيل التفاصيل: عمليات التسجيل، وواجهة برمجة تطبيقات الاستجابات، والتضمينات، وأي الصور، إلى الإعداد، والنسخة الصوتية، تحويل النص إلى كلام، والإشراف، ومفاتيح واجهة برمجة التطبيقات المفقودة. تكامل Cloudflare Quick Tunnel للتواصل مع وكيل السحابي للوصول إليه بعد.![Endpoint Dashboard](screenshots/09-endpoint.png)--- +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) + +--- ## 🔑 API Key Management -إنشاء مفاتيح API ونطاقها لربط وإلها. يمكن أن يكون هناك كل المفاتيح الرئيسية على/موفري خدمات محددة لهم حق الوصول الكامل أو أذونات القراءة فقط. إدارة المفاتيح المرئية مع تكرار الاستخدام.---## 📋 Audit Log +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. -متابعة الإجراءات الإدارية بالتصفية حسب نوع الإجراء والممثل والهدف وعنوان IP والطابع الزمني. سجل الأحداث الأمنية الكاملة.---## 🖥️ Desktop Application +--- -تطبيق Native Electron لسطح المكتب لأنظمة التشغيل Windows وmacOS وLinux. قم بالموافقة على OmniRoute كتطبيق مستقل مع نظام متكامل للنظام والدعم دون الاتصال والتحديث التلقائي والتثبيت بنقرة واحدة. +## 📋 Audit Log -الميزات الرئيسية: +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. -- استقصاء جاهزية الضيوف (لا توجد شاشة عند التشغيل البارد) -- نظام إدارة المنافذ -- اتخاذ القرار بشأن المحتوى -- مثال واحد -- التحديث التلقائي عند إعادة التشغيل -- واجهة المستخدم مشروطة بالكامل (إشارات المرور لنظام التشغيل MacOS، وشريط العنوان الإلكتروني لنظام التشغيل Windows/Linux) -- بناء الإلكترون المقوى - يتم إبتكار "وحدات_العقدة" وتشهد بالرمز في المقترحات ورفضها قبل قبولها، مما يمنع الاعتماد في وقت التشغيل على البناء (الإصدار 2.5.5+) +--- -📖 راجع [`electron/README.md`](../electron/README.md) للحصول على التوثيق الكامل. +## 🖥️ Desktop Application + +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. + +Key features: + +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) + +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/ar/docs/TROUBLESHOOTING.md b/docs/i18n/ar/docs/TROUBLESHOOTING.md index 0d1086b79c..a56950cd98 100644 --- a/docs/i18n/ar/docs/TROUBLESHOOTING.md +++ b/docs/i18n/ar/docs/TROUBLESHOOTING.md @@ -4,65 +4,148 @@ --- -المشاكل والحلول الشائعة لـ OmniRoute.---## Quick Fixes -| مشكلة | الحل | -| ------------------------------------- | --------------------------------------------------------------------- | --------------------- | -| تسجيل الدخول الأول لا يعمل | قم بزيارة `INITIAL_PASSWORD` في `.env` (بدون ترميز افتراضي) | -| بدأت لوحة المعلومات على المنفذ الخاطئ | قم بزيارة `PORT=20128` و`NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| لا توجد سجلات للطلب ضمن `السجلات/` | اضبط `ENABLE_REQUEST_LOGS=true` | -| EACCES: تم رفض الإذن | اضبط `DATA_DIR=/path/to/writable/dir` لتجاوز `~/.omniroute` | -| استراتيجية لا تنقذ | التحديث إلى الإصدار 1.4.11+ (إصلاح مخطط Zod لاستمرارية الإعدادات) | ---## Provider Issues | + +Common problems and solutions for OmniRoute. + +--- + +## Quick Fixes + +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- + +## Provider Issues ### "Language model did not provide messages" -**السبب:**استنفدت حصة الموفر. +**Cause:** Provider quota exhausted. -**الإصلاح:** +**Fix:** -1. تحقق من تعقب الحصص في لوحة القيادة -2. استخدم المجموعة من المستويات التجريبية -3. قم بالبديل إلى اللغة اللاتينية الأرخص/المجانية### تحديد المعدل +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**السبب:**استنفدت حصة الاشتراك. +### Rate Limiting -**الإصلاح:** +**Cause:** Subscription quota exhausted. -- إضافة بيع: `cc/clude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- استخدم GLM/MiniMax كنسخة بيعية بسعر رخيص### OAuth Token منتهي الصلاحية +**Fix:** -يقوم OmniRoute بكتابة الشعارات المميزة. إذا كانت هناك مشاكل: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. لوحة المعلومات → الموفر → إعادة الاتصال -2. قم بإلغاء الحذف وإضافة اتصال الموفر---## Cloud Issues +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- + +## Cloud Issues ### Cloud Sync Errors -1. تحقق من نقاط `BASE_URL` لمثيلك قيد التشغيل (على سبيل المثال، `http://localhost:20128`) -2. تحقق من نقاط `CLOUD_URL` إلى نقطة نهاية السحابة الخاصة بك (على سبيل المثال، `https://omniroute.dev`) -3. حافظ على قيم `NEXT_PUBLIC_*` مع قيم من جانب العمال### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**العلامة:**`الرمز المميز 'd'...' غير متوقع في نقطة نهاية السحابة للمكالمات غير المتدفقة. +### Cloud `stream=false` Returns 500 -**السبب:**يقوم المنبع بإرجاع حمولة SSE أثناء العميل JSON. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**الحل البديل:**استخدم `stream=true` للمكالمات السحابية المباشرة. قم بتضمين SSE المحلي → JSON الاحتياطي.### السحابة تقول أنها متصلة ولكن "مفتاح API غير صالح" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. أنشئ مفتاحًا جديدًا من لوحة التحكم المحلية (`/api/keys`) -2. قم بتشغيل البروتوكولات السحابية: قم بتمكين السحابة → النوبات الآن -3. لا يزال بإمكانها المفاتيح القديمة/غير المتزامنة إرجاع "401" على السحابة---## Docker Issues +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- + +## Docker Issues ### CLI Tool Shows Not Installed -1. تحقق من استهلاك وقت التشغيل: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. بالنسبة لوضع الهاتف المحمول: استخدم الصورة الهدف `runner-cli` (CLIs المجمعة) -3. بالنسبة لصلاحية التثبيت المحلية: قم بـ `CLI_EXTRA_PATHS` وتثبيت دليل المضيفة للقراءة فقط -4. إذا كان "تم التثبيت = صحيح" و"قابل للتشغيل = خطأ": تم العثور على الملف الثنائي ولكن فشل التحقق من الصحة### Quick Runtime Validation```bash - curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' - curl -s http://localhost:20128/api/cli-tools/claude-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' - curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck -```` +### Quick Runtime Validation + +```bash +curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' +curl -s http://localhost:20128/api/cli-tools/claude-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' +curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' +``` --- @@ -70,106 +153,160 @@ ### High Costs -1. التحقق من إحصائيات استخدام لوحة المعلومات → -2. قم باستبدال النموذج الأساسي بـ GLM/MiniMax -3. استخدم طلب التقديم (Gemini CLI، Qoder) للمهام غير المرغوب فيه -4. قم بإنشاء اقتصاديات التكلفة لكل مفتاح برمجة التطبيقات: لوحة المعلومات ← مفاتيح برمجة التطبيقات ← الميزانية---## Debugging +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- + +## Debugging ### Enable Request Logs -قم بزيارة `ENABLE_REQUEST_LOGS=true` في ملف `.env` الخاص بك. تسجيل سجلات ضمن دليل "السجلات/".### التحقق من صحة المزود```bash +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health + +```bash # Health dashboard http://localhost:20128/dashboard/health # API health check curl http://localhost:20128/api/monitoring/health -```` +``` ### Runtime Storage -- الحالة الرئيسية: `${DATA_DIR}/storage.sqlite` (الموفرون، المجموعات، الأسماء المستعارة، المفاتيح، الإعدادات) - -استخدام: جداول SQLite في `storage.sqlite` (`usage_history`، `call_logs`، `proxy_logs`) + اختياري `${DATA_DIR}/log.txt` و`${DATA_DIR}/call_logs/` -- أرشيف الطلب: `/logs/...` (عندما يكون `ENABLE_REQUEST_LOGS=true`)---## Circuit Breaker Issues +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- + +## Circuit Breaker Issues ### Provider stuck in OPEN state -عندما يكون حظر دائرة الموفر مفتوحًا، يتم حظره حتى نهاية فترة التهدئة. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**الإصلاح:** +**Fix:** -1. انتقل إلى**لوحة التحكم ← الإعدادات ← طيران** -2. التحقق من بطاقة القاطع الكهربائي الخاصة بالمزود المتأثر -3. انقر فوق**إعادة تعيين الكل**لمسح جميع القواطع، أو انتظر حتى انتهاء فترة التهدئة -4. التحقق من أن الموفر فعلياً قبل العودة### مقدم الخدمة يستمر في قطع قاطع الدائرة الكهربائية +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -إذا وصلت الخدمة بشكل عام في الحالة المفتوحة: +### Provider keeps tripping the circuit breaker -1. تحقق من**لوحة التحكم ← الصحة ← صحة مقدم الخدمة**نمط المعرفة تبني -2. انتقل إلى**الإعدادات → اختلاف → ملفات تعريف الموفر**وكم حتى حد ما -3. تحقق مما إذا كان الموفر قد قام بتغيير حدود واجهة برمجة التطبيقات (API) أو طلب إعادة المصادقة -4. قم بمراجعة القياس عن بعد لزمن التعرض - قد يتسبب في حدوث التأثيرات الناتجة في سبب واحد بسبب انتهاء المهلة---## Audio Transcription Issues +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- + +## Audio Transcription Issues ### "Unsupported model" error -- تأكد من أنك تستخدم القواعد الصحيحة: `deepgram/nova-3` أو `assemblyai/best` -- تحقق من أن الموفر متصل في**لوحة التحكم ← الموفرون**### يعود النسخ فارغًا أو فاشلًا +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- التحقق من تنسيقات الصوت المدعومة: `mp3`، `wav`، `m4a`، `flac`، `ogg`، `webm` -- التحقق من أن حجم الملف يقع ضمن حدود الموفر (عادةً أقل من 25 ميجابايت) -- التحقق من صلاحية مفتاح API الخاص بالموفر في البطاقة المزودة---## Translator Debugging +### Transcription returns empty or fails -استخدم**لوحة المعلومات → المترجم**لتصحيح المناسب لرغبتك: +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card -| الوضع | متى تستخدم | -| ------------------ | ------------------------------------------------------------------------------ | -------------------------- | -| **ساحة اللعب** | قارن ملفات الإدخال/الإخراج جنباً إلى جنب — لا لصق طلباً فاشلاً لترى كيف ترجمته | -| **اختبار الدردشة** | أرسل الرسائل مباشرة وافحص فاعلية الطلب/الاستجابة الكاملة بما في ذلك الرؤوس | -| **الاختبار** | قم بإنهاء الفقرات المجمعة عبر مجموعات محددة على الترجمات المعطلة | -| **مراقبة حية** | شاهد تدفق الطلبات في التنسيق المطلوب على الترجمة المتقطعة | ### مشكلات التنسيق الشائعة | +--- --**لا تضع علامات التفكير**— تحقق مما إذا كان الموفر المستهدف مقبولاً يمكن توقعه -**استدعاءات مساعدة**— قد توفر بعض الترجمات بشكل فعال لحذف الأشخاص غير المساعدين؛ تحقق في وضع الملعب -**مطالبة النظام مفقودة**— نظام Claude وGemini مع المطالبات المختلفة؛ التحقق من إخراج الترجمة -**ترجع سلسلة SDK أولية أخرى من**- تم إصلاح ذلك في الإصدار 1.1.0: تقوم أداة التأثير الفوري الآن وأنواعها غير الممتازة (`x_groq`، و`usage_breakdown`، وما إلى ذلك) التي فشلت في التحقق من صحة OpenAI SDK Pydantic -**GLM/ERNIE يرفض دور `النظام`**- تم إصلاحه في الإصدار 1.1.0: يقوم بـ«تطبيع الدور التنفيذي برسائل مدمجة في النظام في رسائل المستخدم للنماذج غير المتوافقة» -**لم يتم التعرف على دور "المطور"**- تم إصلاحه في الإصدار 1.1.0: تم تحويله تلقائياً إلى "نظام" لتقديم الخدمات غير التابعة لـ OpenAI -**`json_schema` لا يعمل مع Gemini**— تم إصلاحه في الإصدار 1.1.0: تم الآن تحويل `response_format` إلى `responseMimeType` + `responseSchema` الخاص بـ Gemini---## Resilience Settings +## Translator Debugging + +Use **Dashboard → Translator** to debug format translation issues: + +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | + +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- + +## Resilience Settings ### Auto rate-limit not triggering -- يُطبق حتى يتم التعديل تلقائيًا فقط على لوحة مفاتيح برمجة التطبيقات (وليس OAuth/الاشتراك) -- تحقق من أن**الإعدادات → ← ملفات تعريف الموفر**تم وأيضا تعديلها بشكل تلقائي -- تحقق مما إذا كان الموفر يعرض رموز الحالة "429" أو الذاكرة "إعادة المحاولة بعد".### ضبط التراجع الأسي +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -تدعم ملفات تعريف الموفر هذه الإعدادات: +### Tuning exponential backoff --**التأخير الأساسي**— وقت الانتظار الأول بعد الأول (الافتراضي: 1 ثانية) -**الحد الأقصى للتأخير**— الحد الأقصى لوقت الانتظار (الافتراضي: 30 ثانية) -**المضاعف**— كمية الزيادة الشاملة لكل فشل متتالي (الافتراضي: 2x)### قطيع مضاد الرعد +Provider profiles support these settings: -عندما تصل العديد من الطلبات المتزامنة إلى وفرة السعر، يستخدم تقنية OmniRoute تقنية Mutex + تحديد موعد مباشر لتنزيل الطلبات مباشرة من أجل توقف الحالات المتتالية. وهذا تلقائي لموفري مفاتيح API.---## Optional RAG / LLM failure taxonomy (16 problems) +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) -يقوم بعض مستخدمي OmniRoute بالتحرك أمام RAG أو مكدسات الوكيل. في هذه الإعدادات، من الشائع رؤية نمط غريب: يبدو OmniRoute سليمًا (مقدمو خدمة في وضع جيد، وإصدار الأحكام الشخصية على ما بعد، ولا توجد تنبيهات بخلاف حدود القضاء) ولكن الإجابة لا تزال لا تزال صحيحة. +### Anti-thundering herd -ومن ثم، يأتي هذا الذي يأتي من خط الأنابيب النهائي RAG، وليس من نفسه. +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. -إذا كنت تريد مفردات الحصول على وصف لتلك الإخفاقات، فيمكنك استخدام WFGY IssueMap، وهو مصدر ترخيص MIT الخارجي يحدد ستة عشرة نمطًاًا لفشل RAG / LLM. على مستوى عال يغطي: +--- -- الانجراف استرجاع وحدود السياقة المكسورة -- الفهارس الفارغة أو القديمة ومخازن المتجهات -- التضمين مقابل عدم التطابق الدلالي -- رخص السياقة ورافعة السياقة -- مجموعة واسعة من الإجابات الاستخدام في التجارة الحرة -- خلل في النص بين النص والوكيل -- ذاكرة متعددة للعامل والمؤثرات -- مشاكل النشر والتمهيد +## Optional RAG / LLM failure taxonomy (16 problems) -فكرة بسيطة: +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -1. عندما تقوم بالتحقق من خلل في حسابك، قم بالقاطع: - - مهمة المستخدم وطلبه - - مجموعة الطريق أو المورد في OmniRoute - - أي مؤتمر RAG في المراحل النهائية (المستندات المستردة، وأدوات الأدوات، وما إلى ذلك) -2. قم بتخطيط الحادث لواحد أو من أرقام WFGY IssueMap (`رقم 1`...`رقم 16`). -3. قم بتخزين الرقم في لوحة المعلومات الخاصة بك، أو دليل التشغيل، أو أداة التعقب بجوار سجلات OmniRoute. -4. استخدم صفحة WFGY لتقرر ما إذا كنت تريد تغيير مكدس RAG أو المسترد أو استراتيجية التوجيه. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -النص الكامل والوصفات الملموسة موجودة هنا (ترخيص معهد ماساتشوستس فارس، النص فقط): +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -[الملف التمهيدي لخريطة مشاكل WFGY](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -ستتجاهل هذا القسم إذا لم تسمح لـ RAG أو خطوط الأنابيب الخارجية خلف OmniRoute.---## Still Stuck? +The idea is simple: --**مشكلات GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**الهندسة الداخلية**: راجع [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) للحصول على التفاصيل -**مرجع واجهة برمجة التطبيقات**: راجع [`docs/API_REFERENCE.md`](API_REFERENCE.md) -**لوحة معلومات الصحة**: التحقق من**معلومات اللوحة ← صحة**معرفة النظام في الوقت الفعلي -**المترجم**: استخدم**لوحة المعلومات ← المترجم**ل التصحيح المناسب لك +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. + +Full text and concrete recipes live here (MIT license, text only): + +[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) + +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- + +## Still Stuck? + +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/ar/llm.txt b/docs/i18n/ar/llm.txt new file mode 100644 index 0000000000..e9a26be0e3 --- /dev/null +++ b/docs/i18n/ar/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (العربية) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## نظرة عامة + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### الأمان +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/bg/README.md b/docs/i18n/bg/README.md index ec9265e506..3445919355 100644 --- a/docs/i18n/bg/README.md +++ b/docs/i18n/bg/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Вашият универсален API прокси — една крайна точка, 60+ доставчици, нулев престой. Сега с**MCP сървър (25 инструмента)**,**A2A протокол**,**Системи за памет/умения**и**Electron Desktop App**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Завършвания на чат • Вграждания • Генериране на изображения • Видео • Музика • Аудио • Прекласиране •**Уеб търсене**• MCP сървър • A2A протокол • 100% TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Вашият универсален API прокси — една крайна [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Уебсайт](https://omniroute.online) • [🚀 Бърз старт](#-бърз старт) • [💡 Функции](#-ключови-функции) • [📖 Документи](#-документация) • [💰 Ценообразуване](#-ценообразуване с един поглед) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Налично на:**🇺🇸 [английски](README.md) | 🇧🇷 [Португалски (Бразилия)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [италиански](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Нидерландия](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Португалия)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Полски](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [филипински](docs/i18n/phi/README.md) | 🇨🇿 [Чещина](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,555 +60,629 @@ _Вашият универсален API прокси — една крайна ## 📸 Dashboard Preview -<подробности> +
+Click to see dashboard screenshots -Щракнете, за да видите екранни снимки на таблото за управление +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | -| Страница | Екранна снимка | -| -------------------------- | ----------------------------------------------------- | ---------- | -| **Доставчици** | ![Доставчици](docs/screenshots/01-providers.png) | -| **Комбота** | ![Комбинации](docs/screenshots/02-combos.png) | -| **Анализ** | ![Анализ](docs/screenshots/03-analytics.png) | -| **Здраве** | ![Здраве](docs/screenshots/04-health.png) | -| **Преводач** | ![Преводач](docs/screenshots/05-translator.png) | -| **Настройки** | ![Настройки](docs/screenshots/06-settings.png) | -| **CLI инструменти** | ![CLI инструменти](docs/screenshots/07-cli-tools.png) | -| **Дневници за използване** | ![Използване](docs/screenshots/08-usage.png) | -| **Крайни точки** | ![Крайни точки](docs/screenshots/09-endpoint.png) |
| + --- ### 🤖 Free AI Provider for your favorite coding agents -_Свържете всеки базиран на AI IDE или CLI инструмент чрез OmniRoute — безплатен API шлюз за неограничено кодиране._ - -<таблица> - - - -OpenClaw
-OpenClaw -

-⭐ 205K - - - -NanoBot
-NanoBot -

-⭐ 20,9K - - - -PicoClaw
-PicoClaw -

-⭐ 14,6K - - - -ZeroClaw
-ZeroClaw -

-⭐ 9,9K - - - -IronClaw
-Железен нокът -

-⭐ 2,1K - - - - - -OpenCode
-OpenCode -

-⭐ 106K - - - -Codex CLI
-Codex CLI -

-⭐ 60,8K - - - -Claude Code
-Клод Код -

-⭐ 67,3K - - - -Gemini CLI
-Gemini CLI -

-⭐ 94,7K - - - -Kilo Code
-Код на килограм -

-⭐ 15,5K - - +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ + + + + + + + + + + + + + + +
+ + OpenClaw
+ OpenClaw +

+ ⭐ 205K +
+ + NanoBot
+ NanoBot +

+ ⭐ 20.9K +
+ + PicoClaw
+ PicoClaw +

+ ⭐ 14.6K +
+ + ZeroClaw
+ ZeroClaw +

+ ⭐ 9.9K +
+ + IronClaw
+ IronClaw +

+ ⭐ 2.1K +
+ + OpenCode
+ OpenCode +

+ ⭐ 106K +
+ + Codex CLI
+ Codex CLI +

+ ⭐ 60.8K +
+ + Claude Code
+ Claude Code +

+ ⭐ 67.3K +
+ + Gemini CLI
+ Gemini CLI +

+ ⭐ 94.7K +
+ + Kilo Code
+ Kilo Code +

+ ⭐ 15.5K +
-📡 Всички агенти се свързват чрез http://localhost:20128/v1 или http://cloud.omniroute.online/v1 — една конфигурация, неограничени модели и квота--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Спрете да пилеете пари и да достигате лимити:** +**Stop wasting money and hitting limits:** -- Абонаментната квота изтича неизползвана всеки месец -- Ограниченията на скоростта ви спират да кодирате по средата -- Скъпи API ($20-50/месец на доставчик) -- Ръчно превключване между доставчици +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute решава това:** +**OmniRoute solves this:** -- ✅**Увеличете максимално абонаментите**- Проследете квотата, използвайте всеки бит преди нулиране -- ✅**Автоматичен резервен режим**- Абонамент → API ключ → Евтини → Безплатно, нулев престой -- ✅**Множество акаунти**- Кръгови сметки между акаунти на доставчик -- ✅**Универсален**- Работи с Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, всеки CLI инструмент--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Присъединете се към нашата общност!**[WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Получавайте помощ, споделяйте съвети и бъдете в течение. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Уебсайт**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Проблеми**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Група на общността](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Принос**: Вижте [CONTRIBUTING.md](CONTRIBUTING.md), отворете PR или изберете „добър първи брой“ -**Оригинален проект**: [9router от decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Когато отваряте проблем, моля, изпълнете командата system-info и прикачете генерирания файл:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Това генерира `system-info.txt` с вашата версия на Node.js, версия на OmniRoute, подробности за операционната система, инсталирани CLI инструменти (qoder, gemini, claude, codex, antigravity, droid и т.н.), състояние на Docker/PM2 и системни пакети – всичко, от което се нуждаем, за да възпроизведем бързо проблема ви. Прикачете файла директно към вашия проблем с GitHub.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Всеки разработчик, използващ AI инструменти, се сблъсква с тези проблеми всеки ден.**OmniRoute е създаден, за да разреши всички тях — от преразход на разходите до регионални блокове, от повредени OAuth потоци до операции на протоколи и корпоративна наблюдаемост. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. -<подробности> -💸 1. „Плащам за скъп абонамент, но все още ме прекъсват ограниченията“ +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Разработчиците плащат $20–200/месец за Claude Pro, Codex Pro или GitHub Copilot. Дори и да плащате, квотата има таван — 5 часа използване, седмични лимити или лимити на цените на минута. По средата на сесията на кодиране, доставчикът спира да отговаря и разработчикът губи поток и производителност. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Как OmniRoute го решава:** +**How OmniRoute solves it:** --**Smart 4-Tier Fallback**— Ако квотата за абонамент се изчерпи, автоматично пренасочва към API Key → Евтино → Безплатно с нулева ръчна намеса --**Проследяване на ограниченията на доставчика**— Кешираните моментни снимки на квотата се опресняват по график от страна на сървъра (по подразбиране `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) с ръчно опресняване, налично в потребителския интерфейс --**Поддръжка на множество акаунти**— Множество акаунти на доставчик с автоматичен кръгов режим — когато единият свърши, превключва към следващия --**Персонализирани комбинации**— Резервни вериги с възможност за персонализиране с 9 стратегии за балансиране (приоритетни, претеглени, първо запълване, кръгови, P2C, произволни, най-малко използвани, оптимизирани по отношение на разходите, строго произволни) --**Codex Business Quotas**— Мониторинг на квотите на работното пространство на бизнеса/екипа директно в таблото за управление
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard -<подробности> -🔌 2. „Трябва да използвам няколко доставчика, но всеки има различен API“ + -OpenAI използва един формат, Claude (Anthropic) използва друг, Gemini още един. Ако разработчикът иска да тества модели от различни доставчици или резервен вариант между тях, той трябва да преконфигурира SDK, да промени крайните точки, да се справи с несъвместими формати. Персонализираните доставчици (FriendLI, NIM) имат крайни точки на нестандартен модел. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Как OmniRoute го решава:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Unified Endpoint**— Един `http://localhost:20128/v1` служи като прокси за всички 60+ доставчици --**Превод на формати**— Автоматично и прозрачно: OpenAI ↔ Claude ↔ Gemini ↔ Responses API --**Response Sanitization**— Премахва нестандартните полета (`x_groq`, `usage_breakdown`, `service_tier`), които нарушават OpenAI SDK v1.83+ --**Нормализиране на ролята**— Преобразува `developer` → `system` за доставчици, които не са OpenAI; `система` → `потребител` за GLM/ERNIE --**Think Tag Extraction**— Извлича `` блокове от модели като DeepSeek R1 в стандартизирано `reasoning_content` --**Структуриран изход за Gemini**— `json_schema` → `responseMimeType`/`responseSchema` автоматично преобразуване --**`stream` по подразбиране е `false`**— Подравнява се със спецификацията на OpenAI, като се избягват неочаквани SSE в SDK на Python/Rust/Go
+**How OmniRoute solves it:** -<подробности> -🌐 3. „Моят доставчик на AI блокира моя регион/държава“ +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Доставчици като OpenAI/Codex блокират достъпа от определени географски региони. Потребителите получават грешки като `unsupported_country_region_territory` по време на OAuth и API връзки. Това е особено разочароващо за разработчиците от развиващите се страни. + -**Как OmniRoute го решава:** +
+🌐 3. "My AI provider blocks my region/country" --**3-Level Proxy Config**— Конфигурируем прокси на 3 нива: глобално (цял трафик), на доставчик (само един доставчик) и на връзка/ключ --**Цветно кодирани прокси значки**— Визуални индикатори: 🟢 глобален прокси, 🟡 прокси на доставчик, 🔵 прокси за връзка, винаги показващ IP --**OAuth обмен на токени през прокси**— OAuth потокът също минава през проксито, решавайки `unsupported_country_region_territory` --**Тестове за връзка чрез прокси**— Тестовете за връзка използват конфигурирания прокси (без повече директен байпас) --**SOCKS5 Support**— Пълна SOCKS5 прокси поддръжка за изходящо маршрутизиране --**TLS Fingerprint Spoofing**— подобен на браузър TLS пръстов отпечатък чрез `wreq-js` за заобикаляне на откриването на ботове --**🔏 Съпоставяне на пръстови отпечатъци на CLI**— Пренарежда заглавките и полетата на основния текст, за да съответстват на собствените двоични подписи на CLI, драстично намалявайки риска от маркиране на акаунта. Прокси IP адресът се запазва — получавате едновременно стелт**и**IP маскиране
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. -<подробности> -🆓 4. „Искам да използвам AI за кодиране, но нямам пари“ +**How OmniRoute solves it:** -Не всеки може да плаща $20-200/месец за абонаменти за AI. Студенти, разработчици от развиващи се страни, любители и фрийлансъри се нуждаят от достъп до качествени модели на нулева цена. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Как OmniRoute го решава:** + --**Вградени доставчици на безплатни нива**— Вградена поддръжка за 100% безплатни доставчици: Qoder (5 неограничени модела чрез OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 неограничени модела: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID безплатно), Gemini CLI (180K токена/месец безплатно) --**Ollama Cloud**— Хоствани в облака Ollama модели на `api.ollama.com` с безплатно ниво „Light usage“; използвайте префикса `ollamacloud/<модел>` --**Безплатни само комбинации**— Верига `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/месец с нулев престой --**NVIDIA NIM безплатен достъп**— ~40 RPM dev-вечно безплатен достъп до 70+ модела на build.nvidia.com (преход от кредити към чисти лимити на скоростта) --**Стратегия за оптимизиране на разходите**— Стратегия за маршрутизиране, която автоматично избира най-евтиния наличен доставчик +
+🆓 4. "I want to use AI for coding but I have no money" -<подробности> -🔒 5. „Трябва да защитя своя AI шлюз от неоторизиран достъп“ +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -При излагане на AI шлюз към мрежата (LAN, VPS, Docker), всеки с адреса може да използва токените/квотата на разработчика. Без защита приложните програмни интерфейси (API) са уязвими за злоупотреба, незабавно инжектиране и злоупотреба. +**How OmniRoute solves it:** -**Как OmniRoute го решава:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**API Key Management**— Генериране, ротация и обхват за всеки доставчик със специална страница `/dashboard/api-manager` --**Разрешения на ниво модел**— Ограничете API ключовете до конкретни модели (`openai/*`, шаблони със заместващи символи), с превключвател Разрешаване на всички/Ограничаване --**API Endpoint Protection**— Изискване на ключ за `/v1/models` и блокиране на определени доставчици от списъка --**Auth Guard + CSRF Protection**— Всички маршрути на таблото са защитени с мидълуер `withAuth` + CSRF токени --**Ограничител на скоростта**— Ограничаване на скоростта на IP с конфигурируеми прозорци --**IP Filtering**— Списък с разрешени/списък с блокирани за контрол на достъпа --**Prompt Injection Guard**— Дезинфекция срещу злонамерени бързи модели --**AES-256-GCM криптиране**— Идентификационните данни са криптирани в покой
+ -<подробности> -🛑 6. „Доставчикът ми се срина и загубих потока на кодиране“ +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -Доставчиците на AI могат да станат нестабилни, да върнат грешки 5xx или да достигнат временни лимити на скоростта. Ако разработчикът зависи от един доставчик, той е прекъснат. Без прекъсвачи многократните повторни опити могат да сринат приложението. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Как OmniRoute го решава:** +**How OmniRoute solves it:** --**Прекъсвач за всеки модел**— Автоматично отваряне/затваряне с конфигурируеми прагове и изчакване (затворен/отворен/полуотворен), обхват за всеки модел, за да се избегнат каскадни блокове --**Exponential Backoff**— Прогресивни закъснения при повторен опит --**Anti-Thundering Herd**— Mutex + семафорна защита срещу едновременни повторни бури --**Combo Fallback Chains**— Ако основният доставчик се провали, автоматично преминава през веригата без намеса --**Combo Circuit Breaker**— Автоматично деактивира неуспешни доставчици в рамките на комбинирана верига --**Health Dashboard**— Мониторинг на времето на работа, състояния на прекъсвачи, блокировки, статистика на кеша, латентност на p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest -<подробности> -🔧 7. „Конфигурирането на всеки AI инструмент е досадно и повтарящо се“ + -Разработчиците използват Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Всеки инструмент се нуждае от различна конфигурация (крайна точка на API, ключ, модел). Преконфигурирането при смяна на доставчик или модел е загуба на време. +
+🛑 6. "My provider went down and I lost my coding flow" -**Как OmniRoute го решава:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**CLI Tools Dashboard**— Специална страница с настройка с едно кликване за Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline --**GitHub Copilot Config Generator**— Генерира `chatLanguageModels.json` за VS код с масов избор на модел --**Onboarding Wizard**— Насочвана настройка в 4 стъпки за потребители за първи път --**Една крайна точка, всички модели**— Конфигурирайте `http://localhost:20128/v1` веднъж, достъп до 60+ доставчици
+**How OmniRoute solves it:** -<подробности> -🔑 8. „Управлението на OAuth токени от множество доставчици е истински ад“ +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot — всички използват OAuth 2.0 с изтичащи токени. Разработчиците трябва постоянно да се удостоверяват отново, да се справят с „client_secret липсва“, „redirect_uri_mismatch“ и повреди на отдалечени сървъри. OAuth на LAN/VPS е особено проблематичен. + -**Как OmniRoute го решава:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Auto Token Refresh**— OAuth токените се опресняват във фонов режим преди изтичане --**OAuth 2.0 (PKCE) Вграден**— Автоматичен поток за Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder --**Multi-Account OAuth**— Множество акаунти на доставчик чрез JWT/ID извличане на токени --**OAuth LAN/Remote Fix**— Частно IP откриване за `redirect_uri` + ръчен URL режим за отдалечени сървъри --**OAuth зад Nginx**— Използва `window.location.origin` за обратна прокси съвместимост --**Отдалечено ръководство за OAuth**— Ръководство стъпка по стъпка за идентификационни данни на Google Cloud на VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. -<подробности> -📊 9. „Не знам колко харча или къде“ +**How OmniRoute solves it:** -Разработчиците използват множество платени доставчици, но нямат унифициран поглед върху разходите. Всеки доставчик има собствено табло за таксуване, но няма консолидиран изглед. Неочакваните разходи могат да се натрупат. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Как OmniRoute го решава:** + --**Табло за анализ на разходите**— Проследяване на разходите за токени и управление на бюджета за доставчик --**Бюджетни ограничения за ниво**— Таван на разходите за ниво, което задейства автоматичен резервен вариант --**Конфигурация на ценообразуване за модел**— Конфигурируеми цени за модел --**Статистика на използването на API ключ**— Брой заявки и последно използвано клеймо за всеки ключ --**Табло за управление на анализи**— Статистически карти, диаграма на използването на модела, таблица на доставчика с проценти на успех и закъснение +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" -<подробности> -🐛 10. „Не мога да диагностицирам грешки и проблеми в обажданията с AI“ +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Когато обаждането е неуспешно, разработчикът не знае дали е ограничение на скоростта, изтекъл токен, грешен формат или грешка на доставчика. Фрагментирани регистрационни файлове в различни терминали. Без възможност за наблюдение отстраняването на грешки е метод проба-грешка. +**How OmniRoute solves it:** -**Как OmniRoute го решава:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Табло за управление на унифицирани регистрационни файлове**— 4 раздела: регистрационни файлове за заявки, регистрационни файлове за прокси, регистрационни файлове за одит, конзола --**Console Log Viewer**— Преглед в стил терминал в реално време с цветно кодирани нива, автоматично превъртане, търсене, филтър --**SQLite Proxy Logs**— Постоянни регистрационни файлове, които оцеляват при рестартиране на сървъра --**Translator Playground**— 4 режима за отстраняване на грешки: Playground (превод на формат), Chat Tester (обиколно пътуване), Test Bench (партида), Live Monitor (в реално време) --**Заявка за телеметрия**— p50/p95/p99 латентност + проследяване на X-Request-Id --**Регистриране на базата на файлове с ротация**— регистрационните файлове на приложението се редуват по размер, дни на съхранение и брой архиви; артефактите в регистъра на повикванията се редуват по дни на задържане и брой файлове --**Отчет за системна информация**— `npm run system-info` генерира `system-info.txt` с вашата пълна среда (версия на възел, версия на OmniRoute, OS, CLI инструменти, състояние на Docker/PM2). Прикачете го, когато докладвате за проблеми за незабавно сортиране.
+ -<подробности> -🏗️ 11. „Внедряването и поддържането на шлюза е сложно“ +
+📊 9. "I don't know how much I'm spending or where" -Инсталирането, конфигурирането и поддържането на AI прокси в различни среди (локални, VPS, Docker, облак) е трудоемко. Проблеми като твърдо кодирани пътища, `EACCES` в директории, конфликти на портове и междуплатформени компилации добавят триене. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Как OmniRoute го решава:** +**How OmniRoute solves it:** --**npm global install**— `npm install -g omniroute && omniroute` — готово --**Docker Multi-Platform**— роден AMD64 + ARM64 (Apple Silicon, AWS Graviton, Raspberry Pi) --**Docker Compose Profiles**— `base` (без CLI инструменти) и `cli` (с Claude Code, Codex, OpenClaw) --**Electron Desktop App**— родно приложение за Windows/macOS/Linux със системна област, автоматично стартиране, офлайн режим --**Split-Port Mode**— API и табло за управление на отделни портове за разширени сценарии (обратен прокси, контейнерна мрежа) --**Cloud Sync**— Конфигуриране на синхронизиране между устройства чрез Cloudflare Workers --**DB Backups**— Автоматично архивиране, възстановяване, експортиране и импортиране на всички настройки, с `DISABLE_SQLITE_AUTO_BACKUP` за външно управлявани архиви
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency -<подробности> -🌍 12. „Интерфейсът е само на английски и екипът ми не говори английски“ + -Екипите в неанглоговорящите страни, особено в Латинска Америка, Азия и Европа, се затрудняват с интерфейси само на английски. Езиковите бариери намаляват приемането и увеличават грешките в конфигурацията. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Как OmniRoute го решава:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Dashboard i18n — 30 езика**— Всички 500+ преведени клавиша, включително арабски, български, датски, немски, испански, фински, френски, иврит, хинди, унгарски, индонезийски, италиански, японски, корейски, малайски, холандски, норвежки, полски, португалски (PT/BR), румънски, руски, словашки, шведски, тайландски, украински, виетнамски, китайски, филипински, английски --**RTL Support**— Поддръжка отдясно наляво за арабски и иврит --**Многоезични READMEs**— 30 пълни превода на документация --**Избор на език**— Икона на глобус в заглавката за превключване в реално време
+**How OmniRoute solves it:** -<подробности> -🔄 13. „Имам нужда от повече от чат — имам нужда от вграждания, изображения, аудио“ +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -AI не е просто завършване на чат. Разработчиците трябва да генерират изображения, да транскрибират аудио, да създават вграждания за RAG, да прекласират документи и да модерират съдържание. Всеки API има различна крайна точка и формат. + -**Как OmniRoute го решава:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Вграждания**— `/v1/вграждания` с 6 доставчика и 9+ модела --**Генериране на изображения**— `/v1/images/generations` с 10 доставчика и 20+ модела (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Текст към видео**— `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) и SD WebUI --**Текст към музика**— `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) --**Аудио транскрипция**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Текст-към-говор**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + съществуващи доставчици --**Модерации**— `/v1/moderations` — Проверки за безопасност на съдържанието --**Прекласиране**— `/v1/rerank` — Прекласиране на уместността на документа --**API за отговори**— Пълна поддръжка на `/v1/responses` за Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. -<подробности> -🧪 14. „Нямам начин да тествам и сравнявам качеството между моделите“ +**How OmniRoute solves it:** -Разработчиците искат да знаят кой модел е най-подходящ за техния случай на употреба – код, превод, разсъждения – но ръчното сравняване е бавно. Не съществуват интегрирани инструменти за оценка. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**Как OmniRoute го решава:** + --**Оценки на LLM**— Тестване със златен комплект с 10 предварително заредени случая, обхващащи поздрави, математика, география, генериране на код, съответствие с JSON, превод, маркдаун, отказ за безопасност --**4 стратегии за съвпадение**— `exact`, `contains`, `regex`, `custom` (JS функция) --**Translator Playground Test Bench**— Пакетно тестване с множество входове и очаквани изходи, сравнение между доставчици --**Chat Tester**— Пълно двупосочно пътуване с визуално изобразяване на отговора --**Монитор на живо**— Поток в реално време на всички заявки, преминаващи през проксито +
+🌍 12. "The interface is English-only and my team doesn't speak English" -<подробности> -📈 15. „Трябва да мащабирам, без да губя производителност“ +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -Тъй като обемът на заявките нараства, без кеширане едни и същи въпроси генерират дублиращи се разходи. Без идемпотентност, дубликат иска обработка на отпадъци. Трябва да се спазват ограниченията за тарифите за всеки доставчик. +**How OmniRoute solves it:** -**Как OmniRoute го решава:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Семантичен кеш**— Двуслоен кеш (подпис + семантичен) намалява разходите и забавянето --**Request Idempotency**— 5s прозорец за дедупликация за идентични заявки --**Rate Limit Detection**— RPM на доставчик, минимална разлика и максимално едновременно проследяване --**Редактируеми ограничения на скоростта**— Конфигурируеми настройки по подразбиране в Настройки → Устойчивост с постоянство --**API Key Validation Cache**— 3-степенен кеш за производствена производителност --**Здравно табло с телеметрия**— p50/p95/p99 латентност, статистика на кеша, ъптайм
+ -<подробности> -🤖 16. „Искам да контролирам поведението на модела глобално“ +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Разработчици, които искат всички отговори на конкретен език, със специфичен тон или искат да ограничат токените за мотивиране. Конфигурирането на това във всеки инструмент/заявка е непрактично. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Как OmniRoute го решава:** +**How OmniRoute solves it:** --**Инжектиране на системна подкана**— Глобална подкана, приложена към всички заявки --**Thinking Budget Validation**— Разсъждаващ контрол на разпределението на токени за всяка заявка (преминаване, автоматично, персонализирано, адаптивно) --**9 стратегии за маршрутизиране**— Глобални стратегии, които определят как се разпределят заявките --**Wildcard Router**— моделите `provider/*` маршрутизират динамично към всеки доставчик --**Combo Enable/Disable Toggle**— Превключвайте комбинации директно от таблото за управление --**Превключване на доставчика**— Активирайте/деактивирайте всички връзки за доставчик с едно щракване --**Блокирани доставчици**— Изключете определени доставчици от списъка `/v1/models`
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex -<подробности> -🧰 17. „Имам нужда от MCP инструменти като първокласни продуктови възможности“ + -Много AI шлюзове разкриват MCP само като скрит детайл за изпълнение. Екипите се нуждаят от видим, управляем оперативен слой. +
+🧪 14. "I have no way to test and compare quality across models" -**Как OmniRoute го решава:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP се появява в раздела за навигация на таблото за управление и протокол на крайна точка -- Специализирана страница за управление на MCP с процес, инструменти, обхвати и одит -- Вграден бърз старт за `omniroute --mcp` и включване на клиента
+**How OmniRoute solves it:** -<подробности> -🧠 18. „Имам нужда от A2A оркестрация със синхронизиране + пътеки на задачи за поток“ +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Работните процеси на агентите се нуждаят както от директни отговори, така и от дълготрайно поточно изпълнение с контрол на жизнения цикъл. + -**Как OmniRoute го решава:** +
+📈 15. "I need to scale without losing performance" -- A2A JSON-RPC крайна точка (`POST /a2a`) с `message/send` и `message/stream` -- SSE поточно предаване с разпространение на състоянието на терминала -- API на жизнения цикъл на задачите за „tasks/get“ и „tasks/cancel“.
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. -<подробности> -🛰️ 19. „Имам нужда от истинско състояние на MCP процес, а не от познат статус“ +**How OmniRoute solves it:** -Оперативните екипи трябва да знаят дали MCP действително е жив, а не само дали API е достъпен. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**Как OmniRoute го решава:** + -- Сърдечен файл по време на изпълнение с PID, времеви отпечатъци, транспорт, брой инструменти и режим на обхват -- API за състояние на MCP, комбиниращ сърдечен ритъм + скорошна активност -- Карти за състояние на потребителския интерфейс за свежест на процеса/време на работа/пулс +
+🤖 16. "I want to control model behavior globally" -<подробности> -<резюме>📋 20. „Имам нужда от изпълнение на MCP инструмент с възможност за проверка“ +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Когато инструментите променят конфигурацията или задействат оперативни действия, екипите се нуждаят от криминалистична проследимост. +**How OmniRoute solves it:** -**Как OmniRoute го решава:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- Поддържано от SQLite одитно регистриране за извиквания на MCP инструмент -- Филтрира по инструмент, успех/неуспех, API ключ и пагинация -- Таблица за одит на таблото + статистически крайни точки за автоматизация
+ -<подробности> -🔐 21. „Имам нужда от MCP разрешения с обхват за интеграция“ +
+🧰 17. "I need MCP tools as first-class product capabilities" -Различните клиенти трябва да имат най-малко привилегирован достъп до категории инструменти. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Как OmniRoute го решава:** +**How OmniRoute solves it:** -- 10 гранулирани MCP обхвата за контролиран достъп до инструмента -- Налагане на обхват и видимост в потребителския интерфейс за управление на MCP -- Безопасна поза по подразбиране за оперативни инструменти
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding -<подробности> -⚙️ 22. „Имам нужда от оперативни контроли без пренасочване“ + -Екипите се нуждаят от бързи промени във времето на изпълнение по време на инциденти или разходни събития. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**Как OmniRoute го решава:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Превключете комбо активирането директно от таблото за управление на MCP -- Прилагайте профили на устойчивост от предварително дефинирани пакети с правила -- Нулирайте състоянието на прекъсвача от същия операционен панел
+**How OmniRoute solves it:** -<подробности> -🔄 23. „Имам нужда от видимост и анулиране на жизнения цикъл на задачите A2A на живо“ +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Без видимост на жизнения цикъл инцидентите със задачи стават трудни за сортиране. + -**Как OmniRoute го решава:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Списък със задачи/филтриране по състояние/умение с пагинация -- Разбивка на метаданни, събития и артефакти на задачи -- Крайна точка за анулиране на задача и действие на потребителския интерфейс с потвърждение
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. -<подробности> -🌊 24. „Имам нужда от активни показатели на потока за A2A натоварване“ +**How OmniRoute solves it:** -Поточните работни потоци изискват оперативно вникване в паралелността и живите връзки. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**Как OmniRoute го решава:** + -- Броячи на активни потоци, интегрирани в статуса A2A -- Времево клеймо на последната задача и брой на състоянието -- A2A карти на таблото за наблюдение на операциите в реално време +
+📋 20. "I need auditable MCP tool execution" -<подробности> -🪪 25. „Имам нужда от стандартно откриване на агент за клиенти“ +When tools mutate config or trigger ops actions, teams need forensic traceability. -Външните клиенти и оркестраторите се нуждаят от машинночетими метаданни за включване. +**How OmniRoute solves it:** -**Как OmniRoute го решава:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- Карта на агент, изложена в `/.well-known/agent.json` -- Възможности и умения, показани в потребителския интерфейс за управление -- API за състоянието на A2A включва метаданни за откриване за автоматизация
+ -<подробности> -🧭 26. „Имам нужда от откриваемост на протокола в UX на продукта“ +
+🔐 21. "I need scoped MCP permissions per integration" -Ако потребителите не могат да открият повърхности на протокола, качеството на приемане и поддръжка пада. +Different clients should have least-privilege access to tool categories. -**Как OmniRoute го решава:** +**How OmniRoute solves it:** -- Консолидирана страница**Крайни точки**с раздели за прокси, MCP, A2A и API крайни точки -- Превключва състоянието на вградената услуга (онлайн/офлайн) за MCP и A2A -- Връзки от преглед към специални раздели за управление
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling -<подробности> -🧪 27. „Имам нужда от валидиране на протокол от край до край с реални клиенти“ + -Фалшивите тестове не са достатъчни за валидиране на съвместимостта на протокола преди пускане. +
+⚙️ 22. "I need operational controls without redeploying" -**Как OmniRoute го решава:** +Teams need quick runtime changes during incidents or cost events. -- E2E пакет, който зарежда приложение и използва реален MCP SDK клиентски транспорт -- Клиент A2A тества за потоци откриване, изпращане, поточно предаване, получаване и отмяна -- Кръстосана проверка на твърдения срещу MCP одит и API на A2A задачи
+**How OmniRoute solves it:** -<подробности> -📡 28. „Имам нужда от унифицирана видимост във всички интерфейси“ +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -Разделянето на наблюдаемостта по протокол създава слепи зони и по-дълъг MTTR. + -**Как OmniRoute го решава:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Унифицирани табла за управление/логове/аналитика в един продукт -- Здраве + одит + заявка за телеметрия в OpenAI, MCP и A2A слоеве -- Оперативни API за статус и автоматизация
+Without lifecycle visibility, task incidents become hard to triage. -<подробности> -💼 29. „Имам нужда от една среда за изпълнение за прокси + инструменти + оркестрация на агенти“ +**How OmniRoute solves it:** -Изпълнението на много отделни услуги увеличава оперативните разходи и режимите на отказ. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**Как OmniRoute го решава:** + -- OpenAI-съвместим прокси, MCP сървър и A2A сървър в един стек -- Споделено удостоверяване, устойчивост, съхранение на данни и възможност за наблюдение -- Последователен модел на политика във всички повърхности на взаимодействие +
+🌊 24. "I need active stream metrics for A2A load" -<подробности> -🚀 30. „Трябва да изпращам агентски работни потоци без разрастване на лепен код“ +Streaming workflows require operational insight into concurrency and live connections. -Екипите губят скорост, когато свързват множество ad-hoc услуги и скриптове. +**How OmniRoute solves it:** -**Как OmniRoute го решава:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Единна стратегия за крайни точки за клиенти и агенти -- Вграден потребителски интерфейс за управление на протоколи и пътеки за проверка на дим -- Готови за производство основи (сигурност, регистриране, устойчивост, архивиране)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Playbook A: Увеличете максимално платения абонамент + евтино архивиране**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -609,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Playbook B: Стек за кодиране с нулеви разходи**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Playbook C: 24/7 винаги включена резервна верига**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -632,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Playbook D: Операции на агент с MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Настройте AI кодиране за минути при**$0/месец**. Свържете тези безплатни акаунти и използвайте вградената комбинация**Free Stack**. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Стъпка | Действие | Отключени доставчици | +| Step | Action | Providers Unlocked | | ---- | -------------------------------------------------- | ------------------------------------------------------------------ | -| 1 | Свържете**Kiro**(AWS Builder ID OAuth) | Клод Сонет 4.5, Хайку 4.5 —**неограничен**| -| 2 | Свържете**Qoder**(Google OAuth) | kimi-k2-мислене, qwen3-coder-plus, deepseek-r1... —**неограничен**| -| 3 | Свържете**Qwen**(Код на устройството) | qwen3-coder-plus, qwen3-coder-flash... —**неограничен**| -| 4 | Свържете**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180K/мес безплатно**| -| 5 | `/dashboard/combos` →**Безплатен стек ($0)**шаблон | Кръгово обвързване на всички безплатни доставчици автоматично | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Насочете всяка IDE/CLI към:**`http://localhost:20128/v1` · API ключ: `any-string` · Готово. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Допълнително покритие по избор (също безплатно):**Groq API ключ (30 RPM безплатно), NVIDIA NIM (40 RPM безплатно, 70+ модела), Cerebras (1M tok/ден), LongCat API ключ (50M tokens/ден!), Cloudflare Workers AI (10K Neurons/ден, 50+ модела).## Бърз старт +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Бърз старт ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **pnpm потребители:**Стартирайте `pnpm approve-builds -g` след инсталирането, за да активирате собствените скриптове за изграждане, изисквани от `better-sqlite3` и `@swc/core`: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > -> ```баш +> ```bash > pnpm install -g omniroute -> pnpm approve-builds -g # Изберете всички пакети → одобри +> pnpm approve-builds -g # Select all packages → approve > omniroute > ``` -Таблото за управление се отваря на `http://localhost:20128`, а основният URL адрес на API е `http://localhost:20128/v1`. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Команда | Описание | -| ----------------------- | ------------------------------------------------------------------------ | -| `omniroute` | Стартов сървър (`PORT=20128`, API и таблото за управление на същия порт) | -| `omniroute --порт 3000` | Задайте каноничен/API порт на 3000 | -| `omniroute --mcp` | Стартирайте MCP сървър (stdio транспорт) | -| `omniroute --no-open` | Без автоматично отваряне на браузъра | -| `omniroute --help` | Показване на помощ | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Допълнителен режим на разделен порт:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -За повечето внедрявания се нуждаете само от: +For most deployments, you only need: -| Променлива | По подразбиране | Цел | -| ------------------------ | ----------------------------- | -------------------------------------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600000` | Споделена базова линия за извличане нагоре по веригата, скрити изчаквания на Undici, заявки за пръстови отпечатъци на TLS и изчаквания на заявка/прокси за мост на API | -| `STREAM_IDLE_TIMEOUT_MS` | наследява `REQUEST_TIMEOUT_MS` | Максимална празнина между поточно предаване, преди OmniRoute да прекрати SSE потока | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -Обратната съвместимост се запазва: съществуващите `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` и други променливи за изчакване на слой все още работят и заместват споделената базова линия. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Налични са разширени настройки, ако имате нужда от по-фин контрол:| Променлива | По подразбиране | Цел | -| ---------------------------------------------- | ---------------------------------------------- | -------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | наследява `REQUEST_TIMEOUT_MS` | Общо време за изчакване на заявка нагоре по веригата, използвано от основния сигнал за прекъсване на извличането | -| `FETCH_HEADERS_TIMEOUT_MS` | наследява `FETCH_TIMEOUT_MS` | Времево ограничение Undici за получаване на заглавки на отговор нагоре | -| `FETCH_BODY_TIMEOUT_MS` | наследява `FETCH_TIMEOUT_MS` | Времево ограничение на Undici между частите на тялото нагоре (`0` го деактивира) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Време за изчакване на Undici TCP връзка | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | „4000“ | Времето за изчакване на сокета за неактивен поддържащ живот | -| `TLS_CLIENT_TIMEOUT_MS` | наследява `FETCH_TIMEOUT_MS` | Време за изчакване за TLS заявки за пръстови отпечатъци, направени чрез `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | наследява `REQUEST_TIMEOUT_MS` или `30000` | Време за изчакване за пренасочване на прокси `/v1` от API порт към порт на таблото | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Време за изчакване на входящата заявка на мостовия сървър на API | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Времето за изчакване на входящата заглавка на мостовия сървър на API | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | „5000“ | Изчакване за поддържане на активност на мостовия сървър на API | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | „0“ | Времето за изчакване на неактивност на сокета на мостовия сървър на API (`0` го деактивира) | +Advanced overrides are available if you need finer control: -Ако стартирате OmniRoute зад Nginx, Caddy, Cloudflare или друг обратен прокси, уверете се, че проксито -таймаутите също са по-високи от вашите таймаути за поток/извличане на OmniRoute.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Отворете таблото за управление → `Доставчици` и свържете поне един доставчик (OAuth или API ключ). -2. Отворете таблото за управление → `Крайни точки` и създайте API ключ. -3. (По избор) Отворете таблото за управление → `Комбота` и задайте вашата резервна верига.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Работи с Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode и OpenAI-съвместими SDK.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (за операции, управлявани от инструмент):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -След това свържете вашия MCP клиент през `stdio` и тествайте инструменти като: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (за работни процеси от агент към агент):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -761,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Този пакет валидира реални MCP и A2A клиентски потоци срещу работещо приложение.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -769,14 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` -<подробности> +
+Void Linux (`xbps-src` template) -Void Linux (шаблон `xbps-src`) - -За потребители на Void Linux можете да изградите собствен пакет с помощта на `xbps-src`. Запазете този блок като `srcpkgs/omniroute/template`:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -788,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -796,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -872,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -883,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute е наличен като публично изображение на Docker в [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Бързо бягане:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -893,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**С файл на средата:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Използване на Docker Compose:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Поддръжката на таблото за внедряване на Docker вече включва**Cloudflare Quick Tunnel**с едно щракване на `Табло → Крайни точки`. Първият активира изтеглянията `cloudflare` само когато е необходимо, стартира временен тунел към текущата ви крайна точка `/v1` и показва генерирания URL `https://*.trycloudflare.com/v1` директно под нормалния ви обществен URL адрес. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Бележки: +Notes: -- URL адресите за бърз тунел са временни и се променят след всяко рестартиране. -- Бързите тунели не се възстановяват автоматично след рестартиране на OmniRoute или контейнер. Активирайте ги отново от таблото за управление, когато е необходимо. -- Управляваната инсталация в момента поддържа Linux, macOS и Windows на `x64` / `arm64`. -- Управляваните бързи тунели по подразбиране са HTTP/2 транспорт, за да се избегнат шумни QUIC UDP буферни предупреждения в ограничени контейнерни среди. Задайте `CLOUDFLARED_PROTOCOL=quic` или `auto`, ако искате различен транспорт. -- Изображенията на Docker обединяват системни CA корени и ги предават на управляван `cloudflared`, което избягва грешки в TLS доверието, когато тунелът стартира вътре в контейнера. -- SQLite работи в режим WAL. `docker stop` трябва да бъде позволено да завърши, така че OmniRoute да може да провери последните промени обратно в `storage.sqlite`. -- Пакетът Compose файлове вече задава гратисен период от 40 секунди. Ако стартирате изображението директно, запазете `--stop-timeout 40` (или подобно), така че ръчните спирания да не прекъсват почистването при изключване. -- Задайте `CLOUDFLARED_BIN=/absolute/path/to/cloudflared`, ако искате OmniRoute да използва съществуващ двоичен файл, вместо да изтегля такъв. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Използване на Docker Compose с Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute може да бъде сигурно изложен чрез автоматичното SSL осигуряване на Caddy. Уверете се, че DNS A записът на вашия домейн сочи към IP адреса на вашия сървър.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` - -| Изображение | Етикет | Размер | Описание | +| Image | Tag | Size | Description | | ------------------------ | -------- | ------ | --------------------- | -| `diegosouzapw/omniroute` | `последно` | ~250MB | Най-новата стабилна версия | -| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Текуща версия |--- +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | + +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**НОВО!**OmniRoute вече е наличен като**стандартно настолно приложение**за Windows, macOS и Linux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Стартирайте OmniRoute като самостоятелно настолно приложение — без терминал, без браузър, без интернет, необходим за локалните модели. Базираното на Electron приложение включва: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Собствен прозорец**— Специален прозорец на приложението с интеграция в системната област -- 🔄**Автоматично стартиране**— Стартирайте OmniRoute при влизане в системата -- 🔔**Нативни известия**— Получавайте сигнали за изчерпване на квотата или проблеми с доставчика -- ⚡**Инсталиране с едно кликване**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Офлайн режим**— Работи напълно офлайн с пакетния сървър### Бърз старт +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Бърз старт ```bash # Development mode @@ -982,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -Когато е минимизиран, OmniRoute живее в системната област с бързи действия: +When minimized, OmniRoute lives in your system tray with quick actions: -- Отворете таблото -- Промяна на сървърния порт -- Излезте от приложението +- Open dashboard +- Change server port +- Quit application -📖 Пълна документация: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Ниво | Доставчик | Цена | Нулиране на квота | Най-добро за | -| ------------------ | --------------------------- | -------------------------------- | ----------------------- | ----------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 АБОНАМЕНТ** | Claude Code (Pro) | $20/месец | 5 часа + седмично | Вече сте абонирани | -| | Codex (Plus/Pro) | $20-200/месец | 5 часа + седмично | Потребители на OpenAI | -| | Gemini CLI | **БЕЗПЛАТНО** | 180K/месец + 1K/ден | всички! | -| | Копилот на GitHub | $10-19/месец | Месечно | Потребители на GitHub | -| **🔑 КЛЮЧ ЗА API** | NVIDIA NIM | **БЕЗПЛАТНО**(dev forever) | ~40 RPM | 70+ отворени модела | -| | Мозъци | **БЕЗПЛАТНО**(1M ток/ден) | 60K TPM / 30 RPM | Най-бързият в света | -| | Groq | **БЕЗПЛАТНО**(30 RPM) | 14.4K RPD | Ултра-бърз Llama/Gemma | -| | DeepSeek V3.2 | $0,27/$1,10 за 1M | Няма | Обосновка за най-добра цена/качество | -| | xAI Grok-4 Бърз | **$0,20/$0,50 за 1M**🆕 | Няма | Най-бързо + извикване на инструмент, ултраниско | -| | xAI Grok-4 (стандартен) | $0,20/$1,50 за 1M 🆕 | Няма | Разсъждаващ флагман от xAI | -| | Мистрал | Безплатен пробен период + платен | Ограничена скорост | Европейски AI | -| | OpenRouter | Плащане при използване | Няма | 100+ модела агр. | -| **💰 ЕВТИНО** | GLM-5 (чрез Z.AI) 🆕 | $0,5/1 милион | Ежедневно 10 сутринта | 128K изход, най-новият флагман | -| | GLM-4.7 | $0,6/1 милион | Ежедневно 10 сутринта | Резервно копие на бюджета | -| | MiniMax M2.5 🆕 | $0,3/1M вход | 5-часово търкаляне | Разсъждение + агентски задачи | -| | MiniMax M2.1 | $0,2/1 милион | 5-часово търкаляне | Най-евтиният вариант | -| | Kimi K2.5 (Moonshot API) 🆕 | Плащане при използване | Няма | Директен достъп до API на Moonshot | -| | Кими К2 | $9/месец апартамент | 10 милиона токена/месец | Предвидими разходи | -| **🆓 БЕЗПЛАТНО** | Qoder | **$0** | Неограничен | 5 модела неограничено | -| | Куен | **$0** | Неограничен | 4 модела неограничено | -| | Киро | **$0** | Неограничен | Клод Сонет/Хайку (AWS Builder) | -| | LongCat Flash-Lite 🆕 | **$0**(50M ток/ден 🔥) | 1 RPS | Най-голямата безплатна квота на Земята | -| | Опрашвания AI 🆕 | **$0**(не е необходим ключ) | 1 изискване/15s | GPT-5, Claude, DeepSeek, Llama 4 | -| | Cloudflare Workers AI 🆕 | **$0**(10K неврони/ден) | ~150 повторения/ден | 50+ модела, глобално предимство | -| | Scaleway AI 🆕 | **$0**(общо 1 милион токена) | Ограничена скорост | ЕС/GDPR, Qwen3 235B, Llama 70B | > 🆕**Добавени нови модели (март 2026 г.):**Grok-4 Fast семейство на $0,20/$0,50/M (бенчмарк на 1143ms — 30% по-бързо от Gemini 2.5 Flash), GLM-5 чрез Z.AI с 128K изход, MiniMax M2.5 разсъждения, DeepSeek V3.2 актуализирани цени, Kimi K2.5 чрез Moonshot direct API. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 $0 Combo Stack — Пълната безплатна настройка:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Нулев разход. Никога не спира кодирането.**Конфигурирайте това като едно OmniRoute комбо и всички резервни варианти се случват автоматично – без ръчно превключване.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Всички модели по-долу са**100% безплатни без изискване за кредитна карта**. OmniRoute автоматично пренасочва между тях, когато една квота изтече — комбинирайте ги всички за неразбиваема комбинация от $0.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Модел | Префикс | Лимит | Ограничение на скоростта | +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | --------------------- | -| `claude-sonnet-4.5` | `kr/` |**Неограничен**| Няма отчетено дневно ограничение | -| `claude-haiku-4.5` | `kr/` |**Неограничен**| Няма отчетено дневно ограничение | -| `claude-opus-4.6` | `kr/` |**Неограничен**| Най-новият Opus чрез Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -| Модел | Префикс | Лимит | Ограничение на скоростта | +### 🟢 QODER MODELS (Free PAT via qodercli) + +| Model | Prefix | Limit | Rate Limit | | ------------------ | ------ | ------------- | --------------- | -| `kimi-k2-мислене` | `ако/` |**Неограничен**| Няма отчетено ограничение | -| `qwen3-coder-plus` | `ако/` |**Неограничен**| Няма отчетено ограничение | -| `deepseek-r1` | `ако/` |**Неограничен**| Няма отчетено ограничение | -| `минимакс-m2.1` | `ако/` |**Неограничен**| Няма отчетено ограничение | -| `kimi-k2` | `ако/` |**Неограничен**| Няма отчетено ограничение | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -> Препоръчителен метод за свързване:**Personal Access Token + `qodercli`**. OAuth на браузъра е -> експериментален и деактивиран по подразбиране, освен ако не са конфигурирани променливи на средата `QODER_OAUTH_*`.### 🟡 QWEN MODELS (Device Code Auth) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| Модел | Префикс | Лимит | Ограничение на скоростта | +### 🟡 QWEN MODELS (Device Code Auth) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | ------------------- | -| `qwen3-coder-plus` | `qw/` |**Неограничен**| Няма отчетено ограничение | -| `qwen3-coder-flash` | `qw/` |**Неограничен**| Няма отчетено ограничение | -| `qwen3-coder-next` | `qw/` |**Неограничен**| Няма отчетено ограничение | -| `модел-визия` | `qw/` |**Неограничен**| Мултимодални (изображения) |### 🟣 GEMINI CLI (Google OAuth) +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Модел | Префикс | Лимит | Ограничение на скоростта | -| ------------------------ | ------ | ---------------------------- | ------------- | -| `gemini-3-flash-preview` | `gc/` |**180K tok/месец**+ 1K/ден | Месечно нулиране | -| `gemini-2.5-pro` | `gc/` | 180K/месец (споделен басейн) | Високо качество |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +### 🟣 GEMINI CLI (Google OAuth) -| Ниво | Дневен лимит | Ограничение на скоростта | Бележки | -| ---------- | ------------ | ----------- | ----------------------------------------------------- | -| Безплатно (Dev) | Без ограничение на токена |**~40 RPM**| 70+ модела; преминаване към чисти лимити на лихвите в средата на 2025 г. | +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -Популярни безплатни модели: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) -| Ниво | Дневен лимит | Ограничение на скоростта | Бележки | -| ---- | ----------------- | ---------------- | -------------------------------------------- | -| Безплатно |**1 милион токена/ден**| 60K TPM / 30 RPM | Най-бързият LLM извод в света; нулира ежедневно | +| Tier | Daily Limit | Rate Limit | Notes | +| ---------- | ------------ | ----------- | ------------------------------------------------------ | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -Предлага се безплатно: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -| Ниво | Дневен лимит | Ограничение на скоростта | Бележки | -| ---- | ------------- | ---------------- | ---------------------------------------------- | -| Безплатно |**14,4K RPD**| 30 RPM за модел | Без кредитна карта; 429 на лимит, не се таксува | +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -Предлага се безплатно: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -| Модел | Префикс | Дневна безплатна квота | Бележки | +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` + +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ------------- | ---------------- | ----------------------------------------- | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | + +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` + +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 + +| Model | Prefix | Daily Free Quota | Notes | | ----------------------------- | ------ | ----------------- | ----------------------- | -| `LongCat-Flash-Lite` | `lc/` |**50 милиона токена**💥 | Най-голямата безплатна квота досега | -| `LongCat-Flash-Chat` | `lc/` | 500K токена | Многооборотен чат | -| `LongCat-Flash-Thinking` | `lc/` | 500K токена | Разсъждения / CoT | -| `LongCat-Flash-Thinking-2601` | `lc/` | 500K токена | Версия от януари 2026 г. | -| `LongCat-Flash-Omni-2603` | `lc/` | 500K токена | Мултимодален | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | -> 100% безплатно, докато сте в публична бета версия. Регистрирайте се в [longcat.chat](https://longcat.chat) с имейл или телефон. Нулира всеки ден в 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. -| Модел | Префикс | Ограничение на скоростта | Доставчик зад | +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| `опенай` | `pol/` | 1 изискване/15s | GPT-5 | -| `клод` | `pol/` | 1 изискване/15s | Антропичен Клод | -| `близнаци` | `pol/` | 1 изискване/15s | Google Gemini | -| `deepseek` | `pol/` | 1 изискване/15s | DeepSeek V3 | -| `лама` | `pol/` | 1 изискване/15s | Мета Лама 4 Скаут | -| `мистрал` | `pol/` | 1 изискване/15s | Мистрал AI | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**Нулево триене:**Без регистрация, без API ключ. Добавете доставчика на Опрашвания с празно поле за ключ и той работи веднага.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| Ниво | Ежедневни неврони | Еквивалентно използване | Бележки | -| ---- | ------------- | ----------------------------------------------- | ----------------------- | -| Безплатно |**10 000**| ~150 LLM resp / 500s аудио / 15K вграждания | Global edge, 50+ модела | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 -Популярни безплатни модели: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (безплатно аудио!), `@cf/qwen/qwen2.5-coder-15b-instruct` +| Tier | Daily Neurons | Equivalent Usage | Notes | +| ---- | ------------- | --------------------------------------- | ----------------------- | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -> Изисква API Token + ID на акаунт от [dash.cloudflare.com](https://dash.cloudflare.com). Съхранявайте ID на акаунта в настройките на доставчика.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -| Ниво | Безплатна квота | Местоположение | Бележки | -| ---- | ------------- | ------------ | ---------------------------------- | -| Безплатно |**1M токени**| 🇫🇷 Париж, ЕС | Не е необходима кредитна карта в рамките на лимити | +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -Предлага се безплатно: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 -> Съвместим с ЕС/GDPR. Вземете API ключ на [console.scaleway.com](https://console.scaleway.com). +| Tier | Free Quota | Location | Notes | +| ---- | ------------- | ------------ | ----------------------------------- | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | ->**💡 Най-добрият безплатен стек (11 доставчици, $0 завинаги):** +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` + +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). + +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Киро (kr/) → Клод Сонет/Хайку НЕОГРАНИЧЕНО -> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 НЕОГРАНИЧЕНО -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 милиона токена/ден 🔥 -> Опрашвания (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — не е необходим ключ -> Qwen (qw/) → qwen3-кодер модели НЕОГРАНИЧЕНИ -> Gemini (gemini/) → Gemini 2.5 Flash — 1500 req/ден безплатно -> Cloudflare AI (cf/) → 50+ модела — 10K неврони/ден -> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M безплатни токени (ЕС) -> Groq (groq/) → Llama/Gemma — 14.4K req/ден ултра-бърз -> NVIDIA NIM (nvidia/) → 70+ отворени модела — 40 RPM завинаги -> Cerebras (cerebras/) → Llama/Qwen най-бързият в света — 1M ток/ден -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Транскрибирайте всяко аудио/видео за**$0**— Deepgram води с $200 безплатно, AssemblyAI $50 резервен вариант, Groq Whisper като неограничено аварийно архивиране. +## 🎙️ Free Transcription Combo -| Доставчик | Безплатни кредити | Най-добър модел | Ограничение на скоростта | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. + +| Provider | Free Credits | Best Model | Rate Limit | | ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | -| 🟢**Deepgram**|**$200 безплатно**(регистрация) | `nova-3` — най-добра точност, 30+ езика | Без ограничение на RPM за безплатни кредити | -| 🔵**AssemblyAI**|**$50 безплатно**(регистрация) | `universal-3-pro` — глави, настроение, PII | Без ограничение на RPM за безплатни кредити | -| 🔴**Groq**|**Безплатно завинаги**| `whisper-large-v3` — OpenAI Whisper | 30 RPM (ограничена скорост) | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | -**Предложена комбинация в `/dashboard/combos`:**``` +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -След това в `/dashboard/media` → раздел**Транскрипция**: качете произволен аудио или видео файл → изберете вашата комбинирана крайна точка → получете транскрипция в поддържани формати.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 е създаден като операционна платформа, а не просто релейно прокси.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Характеристика | Какво прави | -| ---------------------------------------------------- | --------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Grok-4 Fast Family** | xAI модели при $0,20/$0,50/M — сравнително време 1143ms (30% по-бързо от Gemini 2.5 Flash) | -| 🧠**GLM-5 чрез Z.AI** | 128K изходен контекст, $0,5/1M — най-новият флагман от семейството GLM | -| 🔮**MiniMax M2.5** | Разсъждение + агентски задачи при $0,30/1M — значително надграждане от M2.1 | -| 🎯**флаг за извикване на инструмент за модел** | `toolCalling: true/false` за модел в системния регистър — AutoCombo пропуска модели без инструмент | -| 🌍**Откриване на многоезични намерения** | PT/ZH/ES/AR ключови думи в точкуването на AutoCombo — по-добър избор на модел за неанглийско съдържание | -| 📊**Резервни резултати, управлявани от бенчмаркове** | Реална p95 латентност от живи заявки емисии комбо точкуване — AutoCombo се учи от действителни данни | -| 🔁**Искайте дедупликация** | Прозорец за дедупиране, базиран на хеш съдържание — безопасен за много агенти, предотвратява дублиране на такси | -| 🔌**Pluggable RouterStrategy** | Разширяем интерфейс `RouterStrategy` — добавете персонализирана логика за маршрутизиране като добавки | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Характеристика | Какво прави | -| ------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Моделна площадка** | Страница на таблото за директно тестване на всеки модел — селектори на доставчик/модел/крайна точка, Monaco Editor, стрийминг, прекъсване, време | -| 🔏**CLI съпоставяне на пръстови отпечатъци** | Подреждане на заглавка/тяло на доставчик, за да съответства на оригиналните CLI подписи — превключете за доставчик в Настройки > Сигурност.**Вашият прокси IP е запазен** | -| 🤝**Поддръжка на ACP (клиентски протокол на агент)** | Откриване на агент на CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw + още 9), генериращ процес, крайна точка `/api/acp/agents` | -| 🤖**Табло за управление на ACP агенти** | Страница за отстраняване на грешки › Агенти — мрежа от 14 агента със статус на инсталиране, версия, персонализирана форма на агент за всеки CLI инструмент. Потребителите на**OpenCode**получават бутон „Изтегляне на opencode.json“, който автоматично генерира готова за използване конфигурация с всички налични модели. | -| 🔧**Маршрутизиране на потребителски модел `apiFormat`** | Персонализираните модели с `apiFormat: "responses"` вече насочват правилно към преводача на API за отговори | -| 🏢**Изолация на работното пространство на Codex** | Множество работни пространства на Codex на имейл — OAuth правилно разделя връзките по ID на работното пространство | -| 🔄**Електронно автоматично актуализиране** | Настолното приложение проверява за актуализации + автоматично инсталиране при рестартиране | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Характеристика | Какво прави | -| --------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**MCP сървър (25 инструмента)** | Инструменти за IDE/агент чрез 3 транспорта: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 ядра + 3 памет + 4 инструмента за умения | -| 🤝**A2A сървър (JSON-RPC + SSE)** | Изпълнение на задачи от агент към агент със синхронизиране и поточно предаване | -| 🧭**Страница с консолидирани крайни точки** | Страница за управление с раздели с раздели Endpoint Proxy, MCP, A2A и API Endpoints | -| 🎚️**Превключватели за активиране/деактивиране на услуги** | Превключватели за ВКЛ./ИЗКЛ. за MCP и A2A с постоянни настройки (по подразбиране: ИЗКЛ.) | -| 🛰️**MCP Runtime Heartbeat** | Реално състояние на процеса (pid, време на работа, възраст на сърдечния ритъм, транспорт, режим на обхвата) | -| 📋**MCP одитна пътека** | Филтрируеми журнали за одит с успех/неуспех и ключово приписване | -| 🔐**Прилагане на обхват на MCP** | 10 подробни разрешения за обхват за контролиран достъп до инструменти | -| 📡**A2A Управление на жизнения цикъл на задачите** | Списък/филтриране на задачи, проверка на събития/артефакти, отмяна на изпълнявани задачи | -| 📋**Откриване на карта на агент** | `/.well-known/agent.json` за автоматично откриване на клиенти | -| 🧪**Протокол E2E Тестова система** | Истински MCP SDK + A2A клиент протича в `test:protocols:e2e` | -| ⚙️**Оперативни контроли** | Превключете комбо, приложете профили на устойчивост, нулирайте прекъсвачите от една контролна повърхност | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Характеристика | Какво прави | -| ---------------------------------------------------------- | ---------------------------------------------------------------------------------------------- | ----------------------- | -| 🎯**Интелигентен 4-степенен резервен вариант** | Автоматичен маршрут: Абонамент → API ключ → Евтини → Безплатно | -| 📊**Проследяване на квоти в реално време** | Брой токени на живо + нулиране на обратното броене на доставчик | -| 🔄**Форматиране на превода** | OpenAI ↔ Claude ↔ Gemini ↔ Отговори с безопасни за схема преобразувания | -| 👥**Поддръжка за множество акаунти** | Няколко акаунта на доставчик с интелигентен избор | -| 🔄**Автоматично опресняване на токени** | OAuth токените се опресняват автоматично с повторен опит | -| 🎨**Персонализирани комбинации** | 9 стратегии за балансиране + резервен контрол на веригата | -| 🌐**Wildcard Router** | `провайдер/*` динамично маршрутизиране | -| 🧠**Мислене за контрол на бюджета** | Лимити за преминаване, автоматични, персонализирани и адаптивни разсъждения | -| 🔀**Псевдоними на модели** | Вграден + персонализиран псевдоним на модела и безопасност на миграцията | -| ⚡**Влошаване на фона** | Насочване на фонови задачи с нисък приоритет към по-евтини модели | -| 🧪**Интелигентно маршрутизиране, съобразено със задачите** | Автоматичен избор на модел по тип съдържание (кодиране/визия/анализ/обобщение) | -| 🔄**A2A Agent Workflows** | Детерминиран оркестратор на FSM за изпълнения на многоетапни агенти със състояние | -| 🔀**Адаптивно маршрутизиране** | Динамична отмяна на стратегия въз основа на обема на токена и сложността на подканата | -| 🎲**Разнообразие от доставчици** | Оценка на ентропията на Шанън за балансиране на разпределението на трафика с автоматично комбо | -| 💬**Системно бързо инжектиране** | Глобални контроли на поведението, прилагани последователно | -| 📄**Съвместимост с API за отговори** | Пълна поддръжка на `/v1/responses` за Codex и разширени агентни работни потоци | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Характеристика | Какво прави | -| ------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**Генериране на изображения** | `/v1/images/generations` с облачен и локален бекенд | -| 📐**Вграждания** | `/v1/embeddings` за търсене и RAG тръбопроводи | -| 🎤**Аудио транскрипция** | `/v1/audio/transcriptions` — 7 доставчика (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), автоматично откриване на език, поддръжка на MP4/MP3/WAV | -| 🔊**Текст към говор** | `/v1/audio/speech` — 10 доставчика (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) с правилни съобщения за грешки | -| 🎬**Видео генериране** | `/v1/videos/generations` (работни процеси ComfyUI + SD WebUI) | -| 🎵**Музикално поколение** | `/v1/music/generations` (работни процеси на ComfyUI) | -| 🛡️**Модерации** | `/v1/moderations` проверки за безопасност | -| 🔀**Прекласиране** | `/v1/rerank` за оценка на уместността | -| 🔍**Търсене в мрежата**🆕 | `/v1/търсене` — 5 доставчика (Serper, Brave, Perplexity, Exa, Tavily), 6 500+ безплатно/месец, автоматичен отказ, кеш | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Характеристика | Какво прави | -| --------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**Прекъсвачи** | За всеки модел пътуване/възстановяване с прагови контроли | -| 🎯**Модели, съобразени с крайни точки** | Персонализираните модели декларират поддържани крайни точки + API формат | -| 🛡️**Anti-Thundering Herd** | Защита на Mutex + семафор при събития за повторен опит/скорост | -| 🧠**Семантичен + кеш на подписа** | Намаляване на разходите/закъснението с два кеш слоя | -| ⚡**Искане на идемпотентност** | Дублиран защитен прозорец | -| 🔒**TLS Fingerprint Spoofing** | Подобен на браузър TLS отпечатък —**намалява откриването на ботове и маркирането на акаунта** | -| 🔏**CLI съпоставяне на пръстови отпечатъци** | Съвпада със собствените подписи на CLI заявка —**намалява риска от забрана, като същевременно запазва IP на проксито** | -| 🌐**IP филтриране** | Списък с разрешени/списъци с блокирани контроли за открити внедрявания | -| 📊**Редактируеми ограничения на скоростта** | Конфигурируеми глобални/на ниво доставчик ограничения с постоянство | -| 📉**Изящна деградация** | Резервни възможности за многослойни възможности, защитаващи основните операции на шлюза | -| 📜**Пътека за одит на конфигурация** | Проследяване на промяна, базирано на разлика, предотвратяващо оперативно отклонение с прости връщания | -| ⏳**Синхронизиране на здравето на доставчика** | Проактивен мониторинг на изтичането на токена, задействащ предупреждения преди неуспешно оторизиране | -| 🚪**Автоматично деактивиране на забранени акаунти** | Оперативен прекъсвач автоматично запечатва трайно блокирани токен акаунти | -| 🔑**API Key Management + Scoping** | Сигурно издаване/ротация на ключове и контроли на модел/доставчик | -| 👁️**Разкриване на API ключ с обхват**🆕 | Възстановяване с включване на API ключове чрез `ALLOW_API_KEY_REVEAL` | -| 🛡️**Защитени `/models`** | Опционално удостоверяване и скриване на доставчик за каталог на модели | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Характеристика | Какво прави | -| --------------------------------------------------------- | --------------------------------------------------------------------------- | ---------------------------- | -| 📝**Заявка + Регистриране на прокси сървър** | Пълно регистриране на заявка/отговор и прокси | -| 📉**Поточно предавани подробни регистрационни файлове**🆕 | Реконструира SSE потоците от полезен товар чисто в потребителския интерфейс | -| 📋**Табло за управление на Unified Logs** | Изгледи на заявка, прокси, одит и конзола на една страница | -| 🔍**Заявка за телеметрия** | p50/p95/p99 латентност и проследяване на заявки | -| 🏥**Здравно табло** | Време на работа, състояния на прекъсване, блокировки, статистика на кеша | -| 💰**Проследяване на разходите** | Контрол на бюджета и видимост на ценообразуването за модел | -| 📈**Аналитични визуализации** | Прозрения за използването на модел/доставчик и изгледи на тенденции | -| 🧪**Рамка за оценка** | Тестване на златен набор с конфигурируеми стратегии за мач | -| 📡**Диагностика на живо**🆕 | Семантичен байпас на кеша за точно комбинирано тестване на живо | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Характеристика | Какво прави | -| ----------------------------------------- | ------------------------------------------------------------------------------ | --------------------- | -| 🌐**Разполагане навсякъде** | Localhost, VPS, Docker, облачни среди | -| 🚇**Cloudflare Tunnel**🆕 | Интеграция с бърз тунел с едно щракване от таблото за управление | -| 🔑**Филтриране на ключови модели на API** | Роден /v1/models отговор, филтриран чрез присвоени контекстни роли на носителя | -| ⚡**Smart Cache Bypass** | Конфигурируеми TTL евристики и контроли за принудително повторно извличане | -| 🔄**Архивиране/Възстановяване** | Експорт/импорт и потоци за възстановяване след бедствие | -| 🧙**Съветник за присъединяване** | Насочвана настройка при първо стартиране | -| 🔧**CLI Tools Dashboard** | Настройка с едно щракване за популярни инструменти за кодиране | -| 🎮**Моделна площадка** | Тествайте всеки доставчик/модел/крайна точка от таблото | -| 🔏**CLI Fingerprint Toggle** | Съвпадение на пръстови отпечатъци за всеки доставчик в Настройки > Сигурност | -| 🌐**i18n (30 езика)** | Пълно табло за управление + езикова поддръжка на документи с RTL покритие | -| 🧹**Изчистване на всички модели** | Изчистване на списък с модели с едно щракване в подробности за доставчика | -| 👁️**Контроли на страничната лента**🆕 | Скриване на компоненти и интеграции от Настройки на външния вид | -| 📋**Шаблони за проблеми** | Стандартизирани GitHub шаблони за грешки и функции | -| 📂**Директория с персонализирани данни** | Замяна на `DATA_DIR` за място за съхранение | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1295,103 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -Когато квотата, скоростта или здравето са неуспешни, OmniRoute автоматично преминава към следващия кандидат без ръчно превключване.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A са откриваеми в UI и документи (не са скрити) -- API за състоянието на протокола разкриват оперативни данни на живо (`/api/mcp/*`, `/api/a2a/*`) -- Таблата за управление включват действия за операции от ден 2 (комбо превключвания, нулиране на прекъсвача, анулиране на задача)#### Translator + validation workflow +#### Protocol management that is visible and operable -Зоната за преводач включва: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Playground**: поискайте проверки за трансформация -**Chat Tester**: пълна заявка/отговор двупосочно -**Тестова стенда**: множество случаи в едно изпълнение -**Монитор на живо**: изглед на трафика в реално време +#### Translator + validation workflow -Плюс проверка на протокола с реални клиенти чрез `npm run test:protocols:e2e`. +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Справка за инструменти, IDE конфигурации и примери за клиенти +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A Server README](src/lib/a2a/README.md)**— Умения, JSON-RPC методи, стрийминг и жизнен цикъл на задачите## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute включва вградена рамка за оценка за тестване на качеството на отговора на LLM спрямо златен набор. Достъп до него чрез**Analytics → Evals**в таблото за управление.### Built-in Golden Set +## 🧪 Evaluations (Evals) -Предварително зареденият "OmniRoute Golden Set" съдържа тестови случаи за: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Поздрави, математика, география, генериране на код -- Съответствие с JSON формат, превод, генериране на маркдаун -- Отказ за безопасност (вредно съдържание), броене, булева логика### Evaluation Strategies +### Built-in Golden Set -| Стратегия | Описание | Пример | -| ------------ | ----------------------------------------------------------------------- | ------------------------------- | --- | -| `точно` | Изходът трябва да съвпада точно | `"4"` | -| `съдържа` | Изходът трябва да съдържа подниз (без значение за малки и големи букви) | `"Париж"` | -| `регекс` | Изходът трябва да съответства на модела на регулярен израз | `"1.*2.*3"` | -| `по поръчка` | Персонализираната JS функция връща true/false | `(изход) => изход.дължина > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) -<подробности> -<резюме>🧩 Настройка на MCP (моделен контекстен протокол) +
+🧩 MCP Setup (Model Context Protocol) -Стартирайте MCP транспорт в режим stdio:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Препоръчителен поток за валидиране: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Свържете вашия MCP клиент през stdio. -2. Стартирайте `omniroute_get_health`. -3. Стартирайте `omniroute_list_combos`. -4. Отворете `/dashboard/mcp`, за да потвърдите пулса, активността и проверката. - -Полезни API за автоматизация: +Useful APIs for automation: - `GET /api/mcp/status` - `GET /api/mcp/tools` - `GET /api/mcp/audit` -- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats` -<подробности> -<резюме>🤝 Настройка на A2A (Agent2Agent) + -Открийте агента:```bash +
+🤝 A2A Setup (Agent2Agent) + +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Изпратете задача:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` - -Управление на жизнения цикъл: +Manage lifecycle: - `GET /api/a2a/status` - `GET /api/a2a/tasks` - `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -Оперативен потребителски интерфейс: +Operational UI: -- `/dashboard/a2a` за видимост на задача/състояние/поток и димни действия
+- `/dashboard/a2a` for task/state/stream observability and smoke actions -<подробности> -<резюме>🧪 Проверка на протокола от край до край + -Валидирайте и двата протокола с реални клиенти:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Това потвърждава: +This verifies: -- MCP SDK клиент за свързване/списък/обаждане -- A2A откриване/изпращане/поток/получаване/отказ -- Кръстосана проверка на данни в MCP одит и API за управление на задачи A2A
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs -<подробности> -<резюме>💳 Доставчици на абонамент### Claude Code (Pro/Max) + + +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1404,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Професионален съвет:**Използвайте Opus за сложни задачи, Sonnet за скорост. OmniRoute проследява квота за модел!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1418,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Всеки акаунт в Codex вече има превключватели на правилата в `Табло за управление -> Доставчици`: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (ВКЛ./ИЗКЛ.): прилага политиката за 5-часов праг на прозореца. -- `Седмично` (ВКЛ./ИЗКЛ.): прилагане на политиката за седмичния праг на прозореца. -- Прагово поведение: когато активиран прозорец достигне >=90% използване, този акаунт се пропуска. -- Ротационно поведение: OmniRoute автоматично пренасочва към следващия отговарящ на условията акаунт в Codex. -- Поведение при нулиране: когато изтече времето за `resetAt` на доставчика, акаунтът отново автоматично става допустим. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Сценарии: +Scenarios: -- `5h ON` + `Weekly ON`: акаунтът се пропуска, когато някой прозорец достигне прага. -- `5h OFF` + `Weekly ON`: само седмично използване може да блокира акаунта. -- `5h ВКЛ.` + `Weekly OFF`: само 5-часово използване може да блокира акаунта. -- `resetAt` премина: акаунтът влиза отново в ротация автоматично (без ръчно повторно активиране).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1443,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Най-добра стойност:**Огромно безплатно ниво! Използвайте това преди платените нива.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1458,71 +1662,91 @@ Models:
-<подробности> -<резюме>🔑 Доставчици на ключове за API### NVIDIA NIM (FREE developer access — 70+ models) +
+🔑 API Key Providers -1. Регистрирайте се: [build.nvidia.com](https://build.nvidia.com) -2. Вземете безплатен API ключ (включени 1000 кредита за изводи) -3. Табло → Добавяне на доставчик → NVIDIA NIM: - - API ключ: `nvapi-вашият-ключ` +### NVIDIA NIM (FREE developer access — 70+ models) -**Модели:**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` и още 50+ +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Професионален съвет:**OpenAI-съвместим API — работи безпроблемно с превода на формати на OmniRoute!### DeepSeek +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -1. Регистрирайте се: [platform.deepseek.com](https://platform.deepseek.com) -2. Вземете API ключ -3. Табло → Добавяне на доставчик → DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -**Модели:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +### DeepSeek -1. Регистрирайте се: [console.groq.com](https://console.groq.com) -2. Вземете API ключ (включено безплатно ниво) -3. Табло → Добавяне на доставчик → Groq +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -**Модели:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Професионален съвет:**Изключително бърз извод — най-добър за кодиране в реално време!### OpenRouter (100+ Models) +### Groq (Free Tier Available!) -1. Регистрирайте се: [openrouter.ai](https://openrouter.ai) -2. Вземете API ключ -3. Табло → Добавяне на доставчик → OpenRouter +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -**Модели:**Достъп до 100+ модела от всички основни доставчици чрез един API ключ. +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Поведение на таблото:**Моделите OpenRouter се управляват от**Налични модели**. Ръчното добавяне, импортиране и автоматично синхронизиране актуализира един и същ списък.
+**Pro Tip:** Ultra-fast inference — best for real-time coding! -<подробности> -<резюме>💰 Евтини доставчици (резервни)### GLM-4.7 (Daily reset, $0.6/1M) +### OpenRouter (100+ Models) -1. Регистрирайте се: [Zhipu AI](https://open.bigmodel.cn/) -2. Вземете API ключ от Coding Plan -3. Табло → Добавяне на API ключ: - - Доставчик: `glm` - - API ключ: `вашият-ключ` +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -**Използвайте:**`glm/glm-4.7` +**Models:** Access 100+ models from all major providers through a single API key. -**Професионален съвет:**Планът за кодиране предлага 3× квота на цена 1/7! Нулирайте всеки ден в 10:00 ч.### MiniMax M2.1 (5h reset, $0.20/1M) +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -1. Регистрирайте се: [MiniMax](https://www.minimax.io/) -2. Вземете API ключ -3. Табло → Добавяне на API ключ + -**Използвайте:**`minimax/MiniMax-M2.1` +
+💰 Cheap Providers (Backup) -**Професионален съвет:**Най-евтината опция за дълъг контекст (1M токени)!### Kimi K2 ($9/month flat) +### GLM-4.7 (Daily reset, $0.6/1M) -1. Абонирайте се: [Moonshot AI](https://platform.moonshot.ai/) -2. Вземете API ключ -3. Табло → Добавяне на API ключ +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**Използвайте:**`kimi/kimi-latest` +**Use:** `glm/glm-4.7` -**Професионален съвет:**Фиксирани $9/месец за 10 милиона токена = $0,90/1 милион ефективна цена!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. -<подробности> -<резюме>🆓 БЕЗПЛАТНИ доставчици (Спешно архивиране)### Qoder (5 FREE models via OAuth) +### MiniMax M2.1 (5h reset, $0.20/1M) + +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `minimax/MiniMax-M2.1` + +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1563,9 +1787,10 @@ Models:
-<подробности> +
+🎨 Create Combos -🎨 Създаване на комбинации### Example 1: Maximize Subscription → Cheap Backup +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1593,8 +1818,10 @@ Cost: $0 forever!
-<подробности> -<резюме>🔧 CLI интеграция### Cursor IDE +
+🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1605,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Използвайте страницата**CLI Tools**в таблото за управление за конфигурация с едно кликване или редактирайте `~/.claude/settings.json` ръчно.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1616,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Вариант 1 — Табло (препоръчително):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Опция 2 — Ръчно:**Редактиране на `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1633,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Забележка:**OpenClaw работи само с локален OmniRoute. Използвайте „127.0.0.1“ вместо „localhost“, за да избегнете проблеми с разрешаването на IPv6.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1647,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Стъпка 1:**Добавете OmniRoute като персонализиран доставчик:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Стъпка 2:**Създайте/редактирайте `opencode.json` в корена на вашия проект:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1673,117 +1909,130 @@ opencode } } } -```` +``` -**Стъпка 3:**Изберете модела в OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Съвет:**Добавете всеки модел, наличен във вашата крайна точка OmniRoute `/v1/models` към раздела `models`. Използвайте формата „провайдер/идентификатор на модел“ от таблото за управление на OmniRoute.
+ --- ## Отстраняване на проблеми -<подробности> -Щракнете, за да разширите ръководството за отстраняване на неизправности +
+Click to expand troubleshooting guide -**„Езиковият модел не предостави съобщения“** +**"Language model did not provide messages"** -- Квотата на доставчика е изчерпана → Проверете инструмента за проследяване на квотата на таблото за управление -- Решение: Използвайте комбо резервен вариант или преминете към по-евтино ниво +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**Ограничаване на скоростта** +**Rate limiting** -- Изчерпване на квотата за абонамент → Резервно връщане към GLM/MiniMax -- Добавете комбо: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**OAuth токенът е изтекъл** +**OAuth token expired** -- Автоматично опресняване от OmniRoute -- Ако проблемите продължават: Табло за управление → Доставчик → Свързване отново +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Високи разходи** +**High costs** -- Проверете статистическите данни за използването в Табло → Разходи -- Превключете основния модел към GLM/MiniMax -- Използвайте безплатно ниво (Gemini CLI, Qoder) за некритични задачи +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Портовете на таблото/API са грешни** +**Dashboard/API ports are wrong** -- `PORT` е каноничният базов порт (и API порт по подразбиране) -- `API_PORT` заменя само OpenAI-съвместим API слушател -- `DASHBOARD_PORT` заменя само слушателя на таблото за управление/Next.js -- Задайте `NEXT_PUBLIC_BASE_URL` на вашето табло за управление/публичен URL (за OAuth обратни извиквания) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Грешки при синхронизиране в облак** +**Cloud sync errors** -- Уверете се, че `BASE_URL` сочи към вашия работещ екземпляр -- Уверете се, че `CLOUD_URL` сочи към вашата очаквана крайна точка в облака -- Поддържайте стойностите на `NEXT_PUBLIC_*` в съответствие със стойностите от страна на сървъра +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Първото влизане не работи** +**First login not working** -- Проверете `INITIAL_PASSWORD` в `.env` -- Ако не е зададена, резервната парола е „123456“. +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Няма регистрационни файлове за заявки** +**No request logs** -- Артефактите на заявката се записват в `DATA_DIR/call_logs/` като един JSON файл на заявка -- Активирайте улавянето на тръбопровода от таблото за управление → Регистри → Искане на регистрационни файлове, ако имате нужда от подробни полезни товари на етап -- Задайте `APP_LOG_TO_FILE=true`, ако също искате регистрационни файлове на конзолата на приложението в `logs/application/app.log` -- Коригирайте `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` и `CALL_LOG_MAX_ENTRIES` според нуждите +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Тестът за връзка показва „Невалидно“ за OpenAI-съвместими доставчици** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Много доставчици не излагат крайна точка `/models` -- OmniRoute v1.0.6+ включва резервно валидиране чрез завършвания на чат -- Уверете се, че основният URL адрес включва суфикс `/v1`### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ Важно за потребители, работещи с OmniRoute на VPS, Docker или друг отдалечен сървър**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -Доставчиците на**Antigravity**и**Gemini CLI**използват**Google OAuth 2.0**. Google изисква „redirect_uri“ в OAuth потока да съвпада точно с един от предварително регистрираните URI адреси в Google Cloud Console на приложението. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -Идентификационните данни за OAuth, включени в OmniRoute, са регистрирани**само за `localhost`**. Когато получите достъп до OmniRoute на отдалечен сървър (напр. `https://omniroute.myserver.com`), Google отхвърля удостоверяването с:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Трябва да създадете**OAuth 2.0 Client ID**в Google Cloud Console с URI на вашия сървър.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Отворете Google Cloud Console** +#### Step-by-step -Отидете на: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Създайте нов OAuth 2.0 клиентски идентификатор** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Щракнете върху**"+ Създаване на идентификационни данни"**→**"OAuth клиентски идентификатор"** -- Тип приложение:**"Уеб приложение"** -- Име: каквото искате (напр. „OmniRoute Remote“) +**2. Create a new OAuth 2.0 Client ID** -**3. Добавете оторизирани URI адреси за пренасочване** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -В полето**„Оторизирани URI адреси за пренасочване“**добавете:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Заменете `your-server.com` с домейна или IP на вашия сървър (включете порта, ако е необходимо, напр. `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. Запазете и копирайте идентификационните данни** +After creating, Google will show the **Client ID** and **Client Secret**. -След създаването Google ще покаже**Клиентски идентификатор**и**Клиентска тайна**. +**5. Set environment variables** -**5. Задайте променливи на средата** +In your `.env` (or Docker environment variables): -Във вашия `.env` (или променливи на средата Docker):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1792,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Рестартирайте OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Опитайте да се свържете отново** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Табло → Доставчици → Antigravity (или Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google вече ще пренасочва правилно към `https://your-server.com/callback`.--- +--- #### Temporary workaround (without custom credentials) -Ако не искате да настроите свои собствени идентификационни данни точно сега, можете да използвате**ръчния URL поток**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute отваря URL адреса за оторизация на Google -2. След упълномощаване Google се опитва да пренасочи към `localhost` (което не успява на отдалечения сървър) -3.**Копирайте пълния URL**от адресната лента на вашия браузър (дори страницата да не се зарежда) -4. Поставете този URL адрес в полето, показано в модала за свързване на OmniRoute -5. Щракнете върху**"Свързване"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Това работи, защото кодът за оторизация в URL адреса е валиден независимо дали страницата за пренасочване е заредена.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. -<подробности> -<резюме>🇧🇷 Versão em Português#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Доставчиците на**Antigravity**и**Gemini CLI**използват**Google OAuth 2.0**за удостоверяване. Google изисква, че `redirect_uri` не използва fluxo OAuth като**exatamente**, за да може URI преди кадастрада да не се използва Google Cloud Console за приложение. +
+🇧🇷 Versão em Português -Като пълномощия за OAuth не се използва OmniRoute в кадастрада**apenas para `localhost`**. Ако имате достъп до OmniRoute в дистанционния сървър (напр.: `https://omniroute.meuservidor.com`), или Google rejeita a autenticação com:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Изпишете точно**OAuth 2.0 Client ID**без Google Cloud Console чрез URI на вашия сървър.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Достъп до Google Cloud Console** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) **2. Crie um novo OAuth 2.0 Client ID** -- Кликнете върху**"+ Създаване на идентификационни данни"**→**"OAuth клиентски идентификатор"** -- Tipo de aplicativo:**"Уеб приложение"** -- Име: escolha qualquer име (напр.: `OmniRoute Remote`) +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Adicione като оторизирани URI адреси за пренасочване** +**3. Adicione as Authorized Redirect URIs** -Без поле**„Оторизирани URI адреси за пренасочване“**, добавете:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` - -> Заменете `seu-servidor.com` домейн или IP на вашия сървър (включително необходим порт, напр.: `http://45.33.32.156:20128/callback`). +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). **4. Salve e copie as credenciais** -Например, Google показва**Клиентски идентификатор**и**Клиентска тайна**. +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -**5. Конфигуриране като variáveis de ambiente** +**5. Configure as variáveis de ambiente** -Не се използва `.env` (или нашите варианти на средата на Docker):```bash +No seu `.env` (ou nas variáveis de ambiente do Docker): + +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1871,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reinicie o OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute - -```` +``` **7. Tente conectar novamente** -Табло → Доставчици → Антигравитация (или Gemini CLI) → OAuth +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Agora или Google пренасочва корретаментно за „https://seu-servidor.com/callback“ и функционира автентичност.--- +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. + +--- #### Workaround temporário (sem configurar credenciais próprias) -Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo**manual de URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. O OmniRoute премахва URL адрес за авторизация от Google -2. Ако не разрешите, пренасочването на Google към „localhost“ (не може да се използва отдалечен сървър) -3.**Копирайте пълния URL адрес**от страницата, която искате да прехвърлите в своя браузър (mesmo que a página não carregue) +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) 4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute -5. Щракнете върху**"Свързване"** +5. Clique em **"Connect"** -> Това заобиколно решение функционира, ако кодът на авторизацията на URL е валиден независимо от пренасочването към пренасочване или не.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1909,64 +2171,73 @@ Se não quiser criar credenciais próprias agora, ainda é possível usar o flux ## 🛠️ Tech Stack -<подробности> -Щракнете, за да разгънете подробностите за технически стек +
+Click to expand tech stack details --**Време на изпълнение**: Node.js 18–22 LTS (⚠️ Node.js 24+**не се поддържа**— собствените бинарни файлове на `better-sqlite3` са несъвместими) --**Език**: TypeScript 5.9 —**100% TypeScript**в `src/` и `open-sse/` (нула `any` в основните модули от v2.0) --**Framework**: Next.js 16 + React 19 + Tailwind CSS 4 --**База данни**: LowDB (JSON) + SQLite (състояние на домейна + регистрационни файлове на прокси + MCP одит + решения за маршрутизиране) --**Схеми**: Zod (валидиране на I/O инструмент за MCP, API договори) --**Протоколи**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Поточно предаване**: Изпратени от сървъра събития (SSE) --**Auth**: OAuth 2.0 (PKCE) + JWT + API ключове + MCP оторизация с обхват --**Тестване**: Node.js тестов инструмент + Vitest (900+ теста, включително модул, интеграция, E2E) --**CI/CD**: Действия на GitHub (автоматично публикуване на npm + Docker Hub при пускане) --**Уебсайт**: [omniroute.online](https://omniroute.online) --**Пакет**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Устойчивост**: прекъсвач, експоненциално отдръпване, анти-гръмотевично стадо, TLS подправяне, автоматично комбинирано самолечение
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Документация -| Документ | Описание | -| ---------------------------------------------- | -------------------------------------------------- | -| [Ръководство на потребителя](docs/USER_GUIDE.md) | Доставчици, комбинации, CLI интеграция, внедряване | -| [Справочник за API](docs/API_REFERENCE.md) | Всички крайни точки с примери | -| [MCP сървър](open-sse/mcp-server/README.md) | 16 MCP инструмента, IDE конфигурации, Python/TS/Go клиенти | -| [A2A сървър](src/lib/a2a/README.md) | JSON-RPC 2.0 протокол, умения, стрийминг, управление на задачи | -| [Auto-Combo Engine](docs/auto-combo.md) | 6-факторно оценяване, пакети с режими, самолечение | -| [Отстраняване на неизправности](docs/TROUBLESHOOTING.md) | Често срещани проблеми и решения | -| [Архитектура](docs/ARCHITECTURE.md) | Системна архитектура и вътрешност | -| [Принос](CONTRIBUTING.md) | Настройка и насоки за разработка | -| [OpenAPI Spec](docs/openapi.yaml) | Спецификация на OpenAPI 3.0 | -| [Правила за сигурност](SECURITY.md) | Отчитане на уязвимости и практики за сигурност | -| [Внедряване на VM](docs/VM_DEPLOYMENT_GUIDE.md) | Пълно ръководство: Настройка на VM + nginx + Cloudflare | -| [Галерия с функции](docs/FEATURES.md) | Визуална обиколка на таблото с екранни снимки | -| [Списък за проверка на изданието](docs/RELEASE_CHECKLIST.md) | Стъпки за валидиране преди пускане |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute има**планирани 210+ функции**в множество фази на разработка. Ето основните области: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Категория | Планирани функции | Акценти | -| ----------------------------- | ---------------- | ----------------------------------------------------------------------------------------------- | -| 🧠**Маршрутизиране и разузнаване**| 25+ | Маршрутизиране с най-ниска латентност, маршрутизиране на базата на етикети, предварителен полет на квота, избор на P2C акаунт | -| 🔒**Сигурност и съответствие**| 20+ | SSRF укрепване, прикриване на идентификационни данни, ограничение на скоростта за крайна точка, обхват на ключ за управление | -| 📊**Наблюдаемост**| 15+ | OpenTelemetry интеграция, мониторинг на квоти в реално време, проследяване на разходите за модел | -| 🔄**Интеграции на доставчици**| 20+ | Регистър на динамичен модел, изчакване на доставчика, Codex за множество акаунти, анализ на квота на Copilot | -| ⚡**Изпълнение**| 15+ | Слой с двоен кеш, кеш за подкани, кеш за отговор, поддържане на активността при поточно предаване, партиден API | -| 🌐**Екосистема**| 10+ | WebSocket API, горещо презареждане на конфигурация, разпределено хранилище за конфигурация, търговски режим |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**OpenCode Integration**— Поддръжка на родния доставчик за IDE за кодиране OpenCode AI -- 🔗**TRAE Integration**— Пълна поддръжка за рамката за разработка на TRAE AI -- 📦**Batch API**— Асинхронна групова обработка за групови заявки -- 🎯**Маршрутизиране на базата на етикети**— Маршрутизирайте заявки въз основа на персонализирани тагове и метаданни -- 💰**Стратегия с най-ниска цена**— Автоматично изберете най-евтиния наличен доставчик +### 🔜 Coming Soon -> 📝 Пълните спецификации на функциите са налични в [`docs/new-features/`](docs/new-features/) (217 подробни спецификации)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1974,18 +2245,20 @@ OmniRoute има**планирани 210+ функции**в множество ### How to Contribute -1. Разклонете хранилището -2. Създайте свой клон на функции (`git checkout -b feature/amazing-feature`) -3. Задайте вашите промени (`git commit -m 'Добавяне на невероятна функция'`) -4. Пуш към клона (`git push origin feature/amazing-feature`) -5. Отворете заявка за изтегляне +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Вижте [CONTRIBUTING.md](CONTRIBUTING.md) за подробни насоки.### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1997,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Специални благодарности на**[9router](https://github.com/decolua/9router)**от**[decolua](https://github.com/decolua)**— оригиналният проект, който вдъхнови това разклонение. OmniRoute се основава на тази невероятна основа с допълнителни функции, мултимодални API и пълно пренаписване на TypeScript. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Специални благодарности на**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— оригиналната реализация на Go, която вдъхнови този JavaScript порт.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Лиценз -Лиценз на MIT - вижте [ЛИЦЕНЗ](ЛИЦЕНЗ) за подробности.--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/bg/docs/ARCHITECTURE.md b/docs/i18n/bg/docs/ARCHITECTURE.md index 0c0398c41f..4bb4e676d5 100644 --- a/docs/i18n/bg/docs/ARCHITECTURE.md +++ b/docs/i18n/bg/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Последна актуализация: 2026-03-28_## Executive Summary -OmniRoute е локален AI маршрутизиращ шлюз и табло за управление, изградено на Next.js. -Той осигурява единична OpenAI-съвместима крайна точка (`/v1/*`) и маршрутизира трафик през множество доставчици нагоре по веригата с превод, резервен вариант, опресняване на токени и проследяване на използването. -Основни възможности: +_Last updated: 2026-03-28_ -- OpenAI-съвместима API повърхност за CLI/инструменти (28 доставчици) -- Превод на заявка/отговор във форматите на доставчика -- Резервна комбинация от модели (последователност от няколко модела) -- Резервен вариант на ниво акаунт (мулти акаунт на доставчик) -- OAuth + API-ключ управление на връзката на доставчика -- Генериране на вграждане чрез `/v1/embeddings` (6 доставчика, 9 модела) -- Генериране на изображения чрез `/v1/images/generations` (4 доставчика, 9 модела) -- Синтактичен анализ на таг за мислене (`...`) за разсъждаващи модели -- Дезинфекция на отговора за стриктна съвместимост с OpenAI SDK -- Нормализиране на ролята (разработчик→система, система→потребител) за съвместимост между доставчици -- Структурирано преобразуване на изход (json_schema → Gemini responseSchema) -- Локална устойчивост за доставчици, ключове, псевдоними, комбинации, настройки, ценообразуване -- Проследяване на използване/разходи и регистриране на заявки -- Допълнителна облачна синхронизация за синхронизиране на множество устройства/състояние -- Списък с разрешени/блокирани IP адреси за контрол на достъпа до API -- Мислещо управление на бюджета (преминаване/автоматично/персонализирано/адаптивно) -- Бързо инжектиране на глобалната система -- Проследяване на сесии и пръстови отпечатъци -- Подобрено ограничаване на скоростта за всеки акаунт със специфични за доставчика профили -- Модел на прекъсвача за устойчивост на доставчика -- Анти-гръмотевична стадна защита с mutex заключване -- Кеш за дедупликация на заявки, базиран на подпис -- Слой на домейна: наличност на модела, правила за разходите, резервна политика, политика за блокиране -- Устойчивост на състоянието на домейна (кеш за запис на SQLite за резервни варианти, бюджети, блокировки, прекъсвачи на верига) -- Механизъм за правила за централизирана оценка на заявката (заключване → бюджет → резервен) -- Заявка за телеметрия с p50/p95/p99 агрегиране на латентност -- ID на корелация (X-Request-Id) за проследяване от край до край -- Регистриране на одит за съответствие с отказ за всеки API ключ -- Eval framework за осигуряване на качеството на LLM -- Resilience UI табло със статус на прекъсвача в реално време -- Модулни OAuth доставчици (12 отделни модула под `src/lib/oauth/providers/`) +## Executive Summary -Основен модел на изпълнение: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- Маршрутите на приложението Next.js под `src/app/api/*` внедряват както API на таблото, така и API за съвместимост -- Споделено SSE/маршрутизиращо ядро в `src/sse/*` + `open-sse/*` обработва изпълнението на доставчика, превода, стрийминг, резервен вариант и използване## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Време за изпълнение на локален шлюз -- API за управление на таблото -- Удостоверяване на доставчика и опресняване на токена -- Заявка за превод и SSE стрийминг -- Локално състояние + постоянство на използване -- Допълнителна синхронизация в облака### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Внедряване на облачна услуга зад `NEXT_PUBLIC_CLOUD_URL` -- SLA/контролна равнина на доставчика извън локалния процес -- Самите външни CLI двоични файлове (Claude CLI, Codex CLI и т.н.)## Dashboard Surface (Current) +### Out of Scope -Главни страници под `src/app/(dashboard)/dashboard/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — бърз старт + преглед на доставчика -- `/dashboard/endpoint` — крайна точка прокси + MCP + A2A + раздели за крайна точка на API -- `/dashboard/providers` — връзки и идентификационни данни на доставчика -- `/dashboard/combos` — комбинирани стратегии, шаблони, правила за маршрутизиране на модели -- `/dashboard/costs` — агрегиране на разходите и видимост на цените -- `/dashboard/analytics` — анализи и оценки на използването -- `/dashboard/limits` — контроли на квоти/ставки -- `/dashboard/cli-tools` — CLI включване, откриване по време на изпълнение, генериране на конфигурация -- `/dashboard/agents` — открити ACP агенти + потребителска регистрация на агент -- `/dashboard/media` — игрище за изображения/видео/музика -- `/dashboard/search-tools` — тестване и история на доставчика на търсене -- `/dashboard/health` — време на работа, прекъсвачи, ограничения на скоростта -- `/dashboard/logs` — регистрационни файлове на заявка/прокси/одит/конзола -- `/dashboard/settings` — раздели за системни настройки (общи, маршрутизиране, комбинирани настройки по подразбиране и т.н.) -- `/dashboard/api-manager` — жизнен цикъл на API ключ и разрешения за модел## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Основни директории: +Main directories: -- `src/app/api/v1/*` и `src/app/api/v1beta/*` за API за съвместимост -- `src/app/api/*` за API за управление/конфигуриране -- Следващото пренаписване в `next.config.mjs` преобразува `/v1/*` в `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Важни пътища за съвместимост: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` — включва потребителски модели с `custom: true` -- `src/app/api/v1/embeddings/route.ts` — генериране на вграждане (6 доставчика) -- `src/app/api/v1/images/generations/route.ts` — генериране на изображения (4+ доставчици, включително Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — специален чат за всеки доставчик -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — специални вграждания за всеки доставчик -- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — специални изображения за всеки доставчик +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` - `src/app/api/v1beta/models/[...path]/route.ts` -Домейни за управление: +Management domains: -- Удостоверяване/настройки: `src/app/api/auth/*`, `src/app/api/settings/*` -- Доставчици/връзки: `src/app/api/providers*` -- Възли на доставчик: `src/app/api/provider-nodes*` -- Персонализирани модели: `src/app/api/provider-models` (GET/POST/DELETE) -- Каталог с модели: `src/app/api/models/route.ts` (GET) -- Прокси конфигурация: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- Ключове/псевдоними/комбота/цени: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- Използване: `src/app/api/usage/*` -- Синхронизиране/облак: `src/app/api/sync/*`, `src/app/api/cloud/*` -- Помощни инструменти за CLI: `src/app/api/cli-tools/*` -- IP филтър: `src/app/api/settings/ip-filter` (GET/PUT) -- Мислен бюджет: `src/app/api/settings/thinking-budget` (GET/PUT) -- Системна подкана: `src/app/api/settings/system-prompt` (GET/PUT) -- Сесии: `src/app/api/sessions` (GET) -- Ограничения на скоростта: `src/app/api/rate-limits` (GET) -- Устойчивост: `src/app/api/resilience` (GET/PATCH) — профили на доставчика, прекъсвач, състояние на ограничение на скоростта -- Нулиране на устойчивостта: `src/app/api/resilience/reset` (POST) — прекъсвачи за нулиране + охлаждане -- Кеш статистики: `src/app/api/cache/stats` (GET/DELETE) -- Наличност на модела: `src/app/api/models/availability` (GET/POST) -- Телеметрия: `src/app/api/telemetry/summary` (GET) -- Бюджет: `src/app/api/usage/budget` (GET/POST) -- Резервни вериги: `src/app/api/fallback/chains` (GET/POST/DELETE) -- Одит на съответствието: `src/app/api/compliance/audit-log` (GET) +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) - Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- Правила: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Policies: `src/app/api/policies` (GET/POST) -Основни модули на потока: +## 2) SSE + Translation Core -- Запис: `src/sse/handlers/chat.ts` -- Оркестрация на ядрото: `open-sse/handlers/chatCore.ts` -- Адаптери за изпълнение на доставчик: `open-sse/executors/*` -- Откриване на формат/конфигурация на доставчик: `open-sse/services/provider.ts` -- Разбор/разрешаване на модела: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Логика за резервен акаунт: `open-sse/services/accountFallback.ts` -- Регистър на преводите: `open-sse/translator/index.ts` -- Трансформации на потока: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- Извличане/нормализиране на използването: `open-sse/utils/usageTracking.ts` -- Анализатор на мислен етикет: `open-sse/utils/thinkTagParser.ts` -- Манипулатор за вграждане: `open-sse/handlers/embeddings.ts` -- Регистър на доставчика на вграждане: `open-sse/config/embeddingRegistry.ts` -- Манипулатор за генериране на изображения: `open-sse/handlers/imageGeneration.ts` -- Регистър на доставчика на изображения: `open-sse/config/imageRegistry.ts` -- Дезинфекция на отговора: `open-sse/handlers/responseSanitizer.ts` -- Нормализация на ролята: `open-sse/services/roleNormalizer.ts` +Main flow modules: -Услуги (бизнес логика): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- Избор/точкуване на акаунт: `open-sse/services/accountSelector.ts` -- Управление на жизнения цикъл на контекста: `open-sse/services/contextManager.ts` -- Налагане на IP филтър: `open-sse/services/ipFilter.ts` -- Проследяване на сесии: `open-sse/services/sessionManager.ts` -- Дедупликация на заявка: `open-sse/services/signatureCache.ts` -- Инжектиране на системна подкана: `open-sse/services/systemPrompt.ts` -- Мислещо управление на бюджета: `open-sse/services/thinkingBudget.ts` -- Маршрутизиране на модел с заместващи символи: `open-sse/services/wildcardRouter.ts` -- Управление на лимита на скоростта: `open-sse/services/rateLimitManager.ts` -- Прекъсвач: `open-sse/services/circuitBreaker.ts` +Services (business logic): -Модули на ниво домейн: +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- Наличност на модела: `src/lib/domain/modelAvailability.ts` -- Правила/бюджети за разходи: `src/lib/domain/costRules.ts` -- Резервна политика: `src/lib/domain/fallbackPolicy.ts` -- Комбо резолвер: `src/lib/domain/comboResolver.ts` -- Политика за блокиране: `src/lib/domain/lockoutPolicy.ts` -- Механизъм за правила: `src/domain/policyEngine.ts` — централизирано блокиране → бюджет → резервна оценка -- Каталог с кодове за грешки: `src/lib/domain/errorCodes.ts` -- ID на заявката: `src/lib/domain/requestId.ts` -- Време за изчакване на извличане: `src/lib/domain/fetchTimeout.ts` -- Заявка за телеметрия: `src/lib/domain/requestTelemetry.ts` -- Съответствие/одит: `src/lib/domain/compliance/index.ts` -- Изпълнител на оценка: `src/lib/domain/evalRunner.ts` -- Устойчивост на състоянието на домейна: `src/lib/db/domainState.ts` — SQLite CRUD за резервни вериги, бюджети, история на разходите, състояние на блокиране, прекъсвачи +Domain layer modules: -Модули за доставчик на OAuth (12 отделни файла под `src/lib/oauth/providers/`): +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -- Индекс на регистъра: `src/lib/oauth/providers/index.ts` -- Индивидуални доставчици: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` -- Тънка обвивка: `src/lib/oauth/providers.ts` — повторно експортиране от отделни модули## 3) Persistence Layer +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -Основно състояние DB (SQLite): +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -- Основна информация: `src/lib/db/core.ts` (better-sqlite3, миграции, WAL) -- Повторно експортиране на фасада: `src/lib/localDb.ts` (тънък слой за съвместимост за повикващите) -- файл: `${DATA_DIR}/storage.sqlite` (или `$XDG_CONFIG_HOME/omniroute/storage.sqlite`, когато е зададено, иначе `~/.omniroute/storage.sqlite`) -- обекти (таблици + KV пространства от имена): providerConnections, providerNodes, modelAliases, комбинации, apiKeys, настройки, ценообразуване,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +## 3) Persistence Layer -Устойчивост на употреба: +Primary state DB (SQLite): -- фасада: `src/lib/usageDb.ts` (декомпозирани модули в `src/lib/usage/*`) -- SQLite таблици в `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` -- незадължителните файлови артефакти остават за съвместимост/отстраняване на грешки (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- наследените JSON файлове се мигрират към SQLite чрез миграции при стартиране, когато има такива +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -DB на състоянието на домейна (SQLite): +Usage persistence: -- `src/lib/db/domainState.ts` — CRUD операции за състояние на домейн -- Таблици (създадени в `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- Модел на кеша за запис: Картите в паметта са авторитетни по време на изпълнение; мутациите се записват синхронно в SQLite; състоянието се възстановява от DB при студен старт## 4) Auth + Security Surfaces +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- Удостоверяване на бисквитките на таблото за управление: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- Генериране/проверка на API ключ: `src/shared/utils/apiKey.ts` -- Тайните на доставчика се запазват в записите `providerConnections` -- Поддръжка на изходящ прокси чрез `open-sse/utils/proxyFetch.ts` (env vars) и `open-sse/utils/networkProxy.ts` (конфигурируем за всеки доставчик или глобален)## 5) Cloud Sync +Domain State DB (SQLite): -- Инициализация на Scheduler: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Периодична задача: `src/shared/services/cloudSyncScheduler.ts` -- Периодична задача: `src/shared/services/modelSyncScheduler.ts` -- Контролен маршрут: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Резервните решения се управляват от `open-sse/services/accountFallback.ts`, като се използват кодове за състояние и евристика за съобщения за грешка. Комбинираното маршрутизиране добавя един допълнителен предпазител: 400-те с обхват на доставчика, като неизправности при блокиране на съдържание нагоре и проверка на роли, се третират като неизправности в локален модел, така че по-късните комбинирани цели все още могат да се изпълняват.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -Опресняването по време на трафик на живо се изпълнява вътре в `open-sse/handlers/chatCore.ts` чрез изпълнител `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -Периодичното синхронизиране се задейства от „CloudSyncScheduler“, когато облакът е активиран.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -Файлове за физическо съхранение: +Physical storage files: -- основна база данни за изпълнение: `${DATA_DIR}/storage.sqlite` -- Редове за заявка: `${DATA_DIR}/log.txt` (компат/дебъг артефакт) -- структурирани архиви на полезния товар на повикванията: `${DATA_DIR}/call_logs/` -- незадължителни сесии за преводач/заявка за отстраняване на грешки: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: API за съвместимост -- `src/app/api/v1/providers/[provider]/*`: специални маршрути за всеки доставчик (чат, вграждания, изображения) -- `src/app/api/providers*`: доставчик CRUD, валидиране, тестване -- `src/app/api/provider-nodes*`: персонализирано съвместимо управление на възли -- `src/app/api/provider-models`: персонализирано управление на модела (CRUD) -- `src/app/api/models/route.ts`: API за каталог на модели (псевдоними + потребителски модели) -- `src/app/api/oauth/*`: потоци OAuth/код на устройство -- `src/app/api/keys*`: жизнен цикъл на местен API ключ -- `src/app/api/models/alias`: управление на псевдоними -- `src/app/api/combos*`: резервно управление на комбо -- `src/app/api/pricing`: отменя ценообразуването за изчисляване на разходите -- `src/app/api/settings/proxy`: конфигурация на прокси (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: тест за свързване на изходящ прокси (POST) -- `src/app/api/usage/*`: API за използване и регистрационни файлове -- `src/app/api/sync/*` + `src/app/api/cloud/*`: облачно синхронизиране и помощници в облака -- `src/app/api/cli-tools/*`: локални CLI конфигурационни писатели/проверки -- `src/app/api/settings/ip-filter`: списък с разрешени/блокирани IP адреси (GET/PUT) -- `src/app/api/settings/thinking-budget`: конфигурация на бюджета на мислещия токен (GET/PUT) -- `src/app/api/settings/system-prompt`: глобална системна подкана (GET/PUT) -- `src/app/api/sessions`: списък на активни сесии (GET) -- `src/app/api/rate-limits`: състояние на ограничение на скоростта за всеки акаунт (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: анализ на заявка, комбо обработка, цикъл за избор на акаунт -- `open-sse/handlers/chatCore.ts`: превод, изпращане на изпълнителя, повторен опит/опресняване, настройка на потока -- `open-sse/executors/*`: специфично за доставчика поведение на мрежа и формат### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: регистър на преводачите и оркестрация -- Заявка за преводачи: `open-sse/translator/request/*` -- Преводачи на отговор: `open-sse/translator/response/*` -- Форматни константи: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: постоянна конфигурация/състояние и устойчивост на домейн на SQLite -- `src/lib/localDb.ts`: повторно експортиране на съвместимост за DB модули -- `src/lib/usageDb.ts`: хронология на използването/фасада на регистрационните файлове на повикванията върху SQLite таблици## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Всеки доставчик има специализиран изпълнител, разширяващ `BaseExecutor` (в `open-sse/executors/base.ts`), който осигурява изграждане на URL адрес, изграждане на заглавка, повторен опит с експоненциално забавяне, кукички за опресняване на идентификационни данни и метода за оркестрация `execute()`. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Изпълнител | Доставчик(и) | Специална обработка | -| ---------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------------------------------------------------- | -| `Изпълнител по подразбиране` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Конфигурация на динамичен URL/заглавие за доставчик | -| `AntigravityExecutor` | Google Антигравитация | Идентификационни номера на персонализирани проекти/сесии, повторен опит след анализ | -| `CodexExecutor` | OpenAI Codex | Вкарва системни инструкции, принуждава усилие за разсъждение | -| `CursorExecutor` | Курсор IDE | ConnectRPC протокол, Protobuf кодиране, подписване на заявка чрез контролна сума | -| `GithubExecutor` | Копилот на GitHub | Опресняване на Copilot token, заглавки, имитиращи VSCode | -| `KiroExecutor` | AWS CodeWhisperer/Киро | AWS EventStream двоичен формат → SSE конвертиране | -| `GeminiCLIExecutor` | Gemini CLI | Цикъл на опресняване на Google OAuth токен | +### Persistence -Всички други доставчици (включително персонализирани съвместими възли) използват `DefaultExecutor`.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Доставчик | Формат | Удостоверяване | Поток | Непоточно | Опресняване на токена | API за използване | -| ----------------- | --------------- | ------------------------------ | ---------------- | --------- | --------------------- | ---------------------------- | ------------------------------ | -| Клод | Клод | API ключ / OAuth | ✅ | ✅ | ✅ | ⚠️ Само администратор | -| Близнаци | близнаци | API ключ / OAuth | ✅ | ✅ | ✅ | ⚠️ Облачна конзола | -| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Облачна конзола | -| Антигравитация | антигравитация | OAuth | ✅ | ✅ | ✅ | ✅ API с пълна квота | -| OpenAI | openai | API ключ | ✅ | ✅ | ❌ | ❌ | -| Кодекс | openai-отговори | OAuth | ✅ принуден | ❌ | ✅ | ✅ Ограничения на скоростта | -| Копилот на GitHub | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Моментни снимки на квоти | -| Курсор | курсор | Персонализирана контролна сума | ✅ | ✅ | ❌ | ❌ | -| Киро | киро | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Ограничения за използване | -| Куен | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ По заявка | -| Qoder | openai | OAuth (основен) | ✅ | ✅ | ✅ | ⚠️ По заявка | -| OpenRouter | openai | API ключ | ✅ | ✅ | ❌ | ❌ | -| GLM/Кими/МиниМакс | Клод | API ключ | ✅ | ✅ | ❌ | ❌ | -| DeepSeek | openai | API ключ | ✅ | ✅ | ❌ | ❌ | -| Groq | openai | API ключ | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | openai | API ключ | ✅ | ✅ | ❌ | ❌ | -| Мистрал | openai | API ключ | ✅ | ✅ | ❌ | ❌ | -| Недоумение | openai | API ключ | ✅ | ✅ | ❌ | ❌ | -| Заедно AI | openai | API ключ | ✅ | ✅ | ❌ | ❌ | -| Фойерверки AI | openai | API ключ | ✅ | ✅ | ❌ | ❌ | -| Мозъци | openai | API ключ | ✅ | ✅ | ❌ | ❌ | -| Cohere | openai | API ключ | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | openai | API ключ | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Откритите изходни формати включват: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. -- `опенай` -- `openaj-отговори` -- „Клод“. -- "близнаци". +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | -Целевите формати включват: +All other providers (including custom compatible nodes) use the `DefaultExecutor`. -- OpenAI чат/Отговори -- Клод -- Gemini/Gemini-CLI/Антигравитационен плик -- Киро -- Курсор +## Provider Compatibility Matrix -Преводите използват**OpenAI като хъб формат**— всички реализации преминават през OpenAI като междинен:``` +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: + +- `openai` +- `openai-responses` +- `claude` +- `gemini` + +Target formats include: + +- OpenAI chat/Responses +- Claude +- Gemini/Gemini-CLI/Antigravity envelope +- Kiro +- Cursor + +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Преводите се избират динамично въз основа на формата на изходния полезен товар и целевия формат на доставчика. +Additional processing layers in the translation pipeline: -Допълнителни слоеве за обработка в тръбопровода за превод: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Дефектиране на отговора**— Премахва нестандартните полета от отговорите във формат OpenAI (както стрийминг, така и без стрийминг), за да се гарантира стриктно съответствие с SDK --**Нормализиране на ролята**— Преобразува `developer` → `system` за цели, които не са OpenAI; обединява `system` → `user` за модели, които отхвърлят системната роля (GLM, ERNIE) --**Извличане на мислен етикет**— Анализира `...` блокове от съдържание в полето `reasoning_content` --**Структуриран изход**— Преобразува OpenAI `response_format.json_schema` в `responseMimeType` + `responseSchema` на Gemini## Supported API Endpoints +## Supported API Endpoints -| Крайна точка | Формат | Манипулатор | -| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------ | -| `POST /v1/chat/completions` | OpenAI чат | `src/sse/handlers/chat.ts` | -| `POST /v1/messages` | Съобщения на Клод | Същият манипулатор (автоматично разпознат) | -| `POST /v1/responses` | OpenAI отговори | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/вграждания` | OpenAI вграждания | `open-sse/handlers/embeddings.ts` | -| `GET /v1/вграждания` | Списък на модели | API маршрут | -| `POST /v1/images/generations` | OpenAI изображения | `open-sse/handlers/imageGeneration.ts` | -| `GET /v1/images/generations` | Списък на модели | API маршрут | -| `POST /v1/providers/{provider}/chat/completions` | OpenAI чат | Специализиран за всеки доставчик с валидиране на модел | -| `POST /v1/providers/{provider}/embeddings` | OpenAI вграждания | Специализиран за всеки доставчик с валидиране на модел | -| `POST /v1/providers/{provider}/images/generations` | OpenAI изображения | Специализиран за всеки доставчик с валидиране на модел | -| `POST /v1/messages/count_tokens` | Клод Токен Брой | API маршрут | -| `GET /v1/models` | Списък с модели на OpenAI | API маршрут (чат + вграждане + изображение + потребителски модели) | -| `GET /api/models/catalog` | Каталог | Всички модели, групирани по доставчик + тип | -| `POST /v1beta/models/*:streamGenerateContent` | Роден Близнаци | API маршрут | -| `GET/PUT/DELETE /api/settings/proxy` | Прокси конфигурация | Конфигурация на мрежов прокси | -| `POST /api/settings/proxy/test` | Прокси свързаност | Крайна точка на теста за изправност/свързване на прокси | -| `GET/POST/DELETE /api/provider-models` | Модели на доставчици | Метаданни за модела на доставчика, поддържащи персонализирани и управлявани налични модели |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -Обходният манипулатор (`open-sse/utils/bypassHandler.ts`) прихваща известни заявки за "изхвърляне" от Claude CLI - пингове за загряване, извличане на заглавия и броене на токени - и връща**фалшив отговор**, без да консумира токени на доставчика нагоре по веригата. Това се задейства само когато `User-Agent` съдържа `claude-cli`.## Request Logger Pipeline +## Bypass Handler -Регистраторът на заявки (`open-sse/utils/requestLogger.ts`) осигурява 7-етапен тръбопровод за регистриране на грешки, деактивиран по подразбиране, активиран чрез `ENABLE_REQUEST_LOGS=true`:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Файловете се записват в `/logs//` за всяка сесия на заявка.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- изчакване на акаунта на доставчика при преходни/скоростни/удостоверителни грешки -- резервен акаунт преди неуспешна заявка -- резервен комбиниран модел, когато пътят на текущия модел/доставчик е изчерпан## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- предварителна проверка и опресняване с повторен опит за опресняващи доставчици -- 401/403 повторен опит след опит за опресняване в основния път## 3) Stream Safety +## 2) Token Expiry -- контролер на потоци с прекъсване на връзката -- поток за превод с промиване в края на потока и обработка на `[DONE]` -- резервна оценка на използването, когато липсват метаданни за използване на доставчика## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- появяват се грешки при синхронизиране, но локалното изпълнение продължава -- планировчикът има логика с възможност за повторен опит, но периодичното изпълнение в момента извиква синхронизиране с един опит по подразбиране## 5) Data Integrity +## 3) Stream Safety -- Миграции на SQLite схема и кукички за автоматично надграждане при стартиране -- наследен JSON → път за съвместимост на миграцията на SQLite## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Източници на видимост по време на изпълнение: +## 4) Cloud Sync Degradation -- регистрационни файлове на конзолата от `src/sse/utils/logger.ts` -- агрегати за използване на заявка в SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- четиристепенно улавяне на подробен полезен товар в SQLite (`request_detail_logs`), когато `settings.detailed_logs_enabled=true` -- текстов регистър на състоянието на заявката в `log.txt` (по избор/compat) -- незадължителни дълбоки регистрационни файлове за заявка/превод под `logs/`, когато `ENABLE_REQUEST_LOGS=true` -- крайни точки за използване на таблото за управление (`/api/usage/*`) за използване на UI +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -Подробно улавяне на полезен товар на заявка съхранява до четири етапа на полезен товар в JSON на маршрутизирано повикване: +## 5) Data Integrity -- необработена заявка, получена от клиента -- преведена заявка, действително изпратена нагоре -- отговор на доставчика, реконструиран като JSON; поточно предаваните отговори се уплътняват до крайното резюме плюс метаданни на потока -- окончателен клиентски отговор, върнат от OmniRoute; поточно предаваните отговори се съхраняват в същата компактна обобщена форма## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- JWT тайна (`JWT_SECRET`) защитава проверката/подписването на бисквитките на сесията на таблото за управление -- Първоначалната парола за зареждане (`INITIAL_PASSWORD`) трябва да бъде изрично конфигурирана за осигуряване при първо стартиране -- API ключ HMAC тайна (`API_KEY_SECRET`) защитава генерирания локален формат на API ключ -- Тайните на доставчика (API ключове/токени) се съхраняват в локалната база данни и трябва да бъдат защитени на ниво файлова система -- Крайните точки за синхронизиране в облак разчитат на удостоверяване на API ключ + семантика на идентификатор на машина## Environment and Runtime Matrix +## Observability and Operational Signals -Променливите на средата, използвани активно от кода: +Runtime visibility sources: -- Приложение/удостоверяване: `JWT_SECRET`, `INITIAL_PASSWORD` -- Съхранение: `DATA_DIR` -- Съвместимо поведение на възел: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Допълнителна отмяна на базата за съхранение (Linux/macOS, когато `DATA_DIR` не е зададено): `XDG_CONFIG_HOME` -- Хеширане на сигурността: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Регистриране: `ENABLE_REQUEST_LOGS` -- Синхронизиране/облачно URL адресиране: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- Изходящ прокси: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` и варианти с малки букви -- Флагове на функцията SOCKS5: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- Помощници за платформа/време на изпълнение (не специфична за приложението конфигурация): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` и `localDb` споделят една и съща основна политика за директория (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) с мигриране на наследени файлове. -2. `/api/v1/route.ts` делегира на същия унифициран конструктор на каталог, използван от `/api/v1/models` (`src/app/api/v1/models/catalog.ts`), за да се избегне семантично отклонение. -3. Request logger записва пълни заглавки/тяло, когато е разрешено; третира регистрационната директория като чувствителна. -4. Поведението в облака зависи от правилния `NEXT_PUBLIC_BASE_URL` и достъпността на крайната точка на облака. -5. Директорията `open-sse/` се публикува като `@omniroute/open-sse`**npm workspace package**. Изходният код го импортира чрез `@omniroute/open-sse/...` (разрешено от Next.js `transpilePackages`). Пътищата на файловете в този документ все още използват името на директорията `open-sse/` за последователност. -6. Диаграмите в таблото за управление използват**Recharts**(базирани на SVG) за достъпни, интерактивни аналитични визуализации (стълбовидни диаграми на използването на модела, таблици с разбивка на доставчиците с проценти на успех). -7. E2E тестовете използват**Playwright**(`tests/e2e/`), изпълняват се чрез `npm run test:e2e`. Единичните тестове използват**Node.js test runner**(`tests/unit/`), изпълняват се чрез `npm run test:unit`. Изходният код под `src/` е**TypeScript**(`.ts`/`.tsx`); работното пространство `open-sse/` остава JavaScript (`.js`). -8. Страницата с настройки е организирана в 5 раздела: Сигурност, Маршрутизация (6 глобални стратегии: първо запълване, кръгова система, p2c, произволна, най-малко използвана, оптимизирана по отношение на разходите), Устойчивост (ограничения на скоростта за редактиране, прекъсвач, политики), AI (мислещ бюджет, системна подкана, кеш за подкана), Разширени (прокси).## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- Изграждане от източник: `npm run build` -- Изграждане на Docker изображение: `docker build -t omniroute .` -- Стартирайте услугата и проверете: +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: - `GET /api/settings` - `GET /api/v1/models` -- CLI целевият базов URL трябва да бъде `http://:20128/v1`, когато `PORT=20128` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/bg/docs/FEATURES.md b/docs/i18n/bg/docs/FEATURES.md index f337763114..a1f923db97 100644 --- a/docs/i18n/bg/docs/FEATURES.md +++ b/docs/i18n/bg/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Визуално ръководство за всеки раздел на таблото за управление OmniRoute.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Управлявайте връзките на доставчици на AI: OAuth доставчици (Claude Code, Codex, Gemini CLI), доставчици на API ключове (Groq, DeepSeek, OpenRouter) и безплатни доставчици (Qoder, Qwen, Kiro). Сметките в Kiro включват проследяване на кредитния баланс — оставащи кредити, обща надбавка и дата на подновяване, видими в Табло за управление → Използване.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Създавайте комбинации за маршрутизиране на модели с 6 стратегии: приоритетни, претеглени, кръгови, произволни, най-малко използвани и оптимизирани по отношение на разходите. Всяка комбинация свързва няколко модела с автоматичен резервен вариант и включва бързи шаблони и проверки за готовност.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Изчерпателни анализи на използването с потребление на токени, оценки на разходите, топлинни карти на активността, седмични диаграми на разпределение и разбивки по доставчик.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Мониторинг в реално време: време на работа, памет, версия, процентили на латентност (p50/p95/p99), статистика на кеша и състояния на прекъсвача на доставчика.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Четири режима за отстраняване на грешки в API преводи:**Playground**(конвертор на формати),**Chat Tester**(заявки на живо),**Test Bench**(пакетни тестове) и**Live Monitor**(поток в реално време).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Тествайте всеки модел директно от таблото. Изберете доставчик, модел и крайна точка, пишете подкани с Monaco Editor, предавайте отговори в реално време, прекъсвайте по средата на потока и преглеждайте показатели за времето.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Цветови теми с възможност за персонализиране за цялото табло. Изберете от 7 предварително зададени цвята (корал, син, червен, зелен, виолетов, оранжев, циан) или създайте персонализирана тема, като изберете всеки шестнадесетичен цвят. Поддържа светъл, тъмен и системен режим.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Изчерпателен панел с настройки с раздели: +Comprehensive settings panel with tabs: --**Общи**— Системно съхранение, управление на архивиране (база данни за експорт/импорт) -**Външен вид**— Селектор на тема (тъмно/светло/система), предварително зададени цветови теми и персонализирани цветове, видимост на журнала за здраве, контроли за видимост на елементи от страничната лента -**Сигурност**— API защита на крайната точка, персонализирано блокиране на доставчика, IP филтриране, информация за сесията -**Маршрутизиране**— Псевдоними на модела, влошаване на фоновата задача -**Устойчивост**— Устойчивост на лимита на скоростта, настройка на прекъсвача, автоматично деактивиране на забранени акаунти, наблюдение на изтичане на доставчика -**Разширени**— Замени на конфигурацията, одитна пътека на конфигурацията, резервен режим на влошаване![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Конфигурация с едно кликване за инструменти за кодиране на AI: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor и Factory Droid. Включва автоматизирано прилагане/нулиране на конфигурация, профили на свързване и картографиране на модела.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Табло за откриване и управление на CLI агенти. Показва мрежа от 14 вградени агента (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) с: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Състояние на инсталацията**— Инсталирано / Не е намерено с откриване на версия -**Протоколни значки**— stdio, HTTP и др. -**Персонализирани агенти**— Регистрирайте всеки CLI инструмент чрез формуляр (име, двоичен файл, команда за версия, аргументи за генериране) -**CLI съпоставяне на пръстови отпечатъци**— Превключване за всеки доставчик, за да съответства на собствените подписи на CLI заявка, намалявайки риска от забрана, като същевременно запазва прокси IP--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Генерирайте изображения, видеоклипове и музика от таблото за управление. Поддържа OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open и MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Регистриране на заявки в реално време с филтриране по доставчик, модел, акаунт и API ключ. Показва кодове за състояние, използване на токени, латентност и подробности за отговора.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Вашата унифицирана крайна точка на API с разбивка на възможностите: завършвания на чатове, API за отговори, вграждания, генериране на изображения, прекласиране, аудио транскрипция, текст към говор, модериране и регистрирани ключове за API. Интегриране на Cloudflare Quick Tunnel и поддръжка на облачен прокси за отдалечен достъп.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -Създаване, обхват и отмяна на API ключове. Всеки ключ може да бъде ограничен до конкретни модели/доставчици с пълен достъп или разрешения само за четене. Визуално управление на ключове с проследяване на използването.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Проследяване на административни действия с филтриране по тип действие, актьор, цел, IP адрес и клеймо за време. Пълна история на събитията за сигурност.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Настолно приложение Native Electron за Windows, macOS и Linux. Стартирайте OmniRoute като самостоятелно приложение с интеграция в системната област, офлайн поддръжка, автоматично актуализиране и инсталиране с едно щракване. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Ключови характеристики: +Key features: -- Проучване на готовността на сървъра (без празен екран при студен старт) -- Системна област с управление на портове -- Политика за сигурност на съдържанието -- Еднократно заключване -- Автоматична актуализация при рестартиране -- Платформено условен потребителски интерфейс (светофари на MacOS, заглавна лента по подразбиране на Windows/Linux) -- Hardened Electron build packaging — символично свързаните `node_modules` в самостоятелния пакет се откриват и отхвърлят преди опаковането, предотвратявайки зависимостта по време на изпълнение от машината за изграждане (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Вижте [`electron/README.md`](../electron/README.md) за пълна документация. +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/bg/docs/TROUBLESHOOTING.md b/docs/i18n/bg/docs/TROUBLESHOOTING.md index e339228624..45718fe147 100644 --- a/docs/i18n/bg/docs/TROUBLESHOOTING.md +++ b/docs/i18n/bg/docs/TROUBLESHOOTING.md @@ -4,65 +4,148 @@ --- -Често срещани проблеми и решения за OmniRoute.---## Quick Fixes -| Проблем | Решение | -| ----------------------------------------------- | ------------------------------------------------------------------------------ | --------------------- | -| Първото влизане не работи | Задайте `INITIAL_PASSWORD` в `.env` (без твърдо кодирано подразбиране) | -| Таблото се отваря на грешен порт | Задайте `PORT=20128` и `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Няма регистрирани файлове за заявки под `logs/` | Задайте `ENABLE_REQUEST_LOGS=true` | -| EACCES: разрешението е показано | Задайте `DATA_DIR=/path/to/writable/dir` да замените `~/.omniroute` | -| Стратегията за маршрутизиране не се запазва | Актуализация до v1.4.11+ (корекция на Zod схема за постоянство на настройките) | ---## Provider Issues | + +Common problems and solutions for OmniRoute. + +--- + +## Quick Fixes + +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- + +## Provider Issues ### "Language model did not provide messages" -**Причина:**Квотата на доставчика е изчерпана. +**Cause:** Provider quota exhausted. -**Коригиране:** +**Fix:** -1. Проверете инструмента за проследяване на квотите на таблото за управление -2. Използвайте комбо с резервни нива -3. Преминете към по-евтино/безплатно ниво### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Причина:**Абонаментната квота е изчерпана. +### Rate Limiting -**Коригиране:** +**Cause:** Subscription quota exhausted. -- Добавете резервен вариант: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Използвайте GLM/MiniMax като евтино резервно копие### OAuth Token Expired +**Fix:** -OmniRoute автоматично опреснява токените. Ако проблемите продължават: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Табло → Доставчик → Свързване отново -2. Изтрийте и добавете отново връзката с доставчика---## Cloud Issues +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- + +## Cloud Issues ### Cloud Sync Errors -1. Проверете дали `BASE_URL` е към вашия работен екземпляр (напр. `http://localhost:20128`) -2. Проверете дали `CLOUD_URL` е към вашата крайна точка в облака (напр. `https://omniroute.dev`) -3. Поддържайте стойността `NEXT_PUBLIC_*` в съответствие със стойността от страната на сървъра### Cloud `stream=false` Връща 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Симптом:**`Неочакван токен 'd'...` в крайната точка на облака за повикане без точно предаване. +### Cloud `stream=false` Returns 500 -**Причина:**Upstream връща SSE полезен продукт, докато клиентът очаква JSON. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Заобиколно решение:**Използвайте `stream=true` за директни повикания в облака. Локалното време за изпълнение включва резервен SSE→JSON.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Създайте нов ключ от локалното табло за управление (`/api/keys`) -2. Стартирайте облачна синхронизация: Активирайте облака → Синхронизирай сега -3. Старите/несинхронизираните ключове все още могат да връщат „401“ в облака---## Docker Issues +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- + +## Docker Issues ### CLI Tool Shows Not Installed -1. Проверете полетата за изпълнение: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. За преносим режим: използвайте целево изображение `runner-cli` (пакетни CLI) -3. За режим на монтиране на хост: задайте `CLI_EXTRA_PATHS` и монтирайте директорията bin на хоста като само за четене -4. Ако `installed=true` и `runnable=false`: двоичният файл е намерен, но проверката на състоянието е неуспешна### Quick Runtime Validation```bash - curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' - curl -s http://localhost:20128/api/cli-tools/claude-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' - curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck -```` +### Quick Runtime Validation + +```bash +curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' +curl -s http://localhost:20128/api/cli-tools/claude-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' +curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' +``` --- @@ -70,108 +153,160 @@ OmniRoute автоматично опреснява токените. Ако п ### High Costs -1. Проверете статистическите данни за употреба в Табло → Използване -2. Превключете основния модел на GLM/MiniMax -3. Използвайте безплатно ниво (Gemini CLI, Qoder) за некритични задачи -4. Задайте бюджети за разходи за API ключ: Табло за управление → API ключове → Бюджет---## Debugging +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- + +## Debugging ### Enable Request Logs -Задайте `ENABLE_REQUEST_LOGS=true` във вашия `.env` файл. Дневниците се появяват в директорията `logs/`.### Проверете здравето на доставчика```bash +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health + +```bash # Health dashboard http://localhost:20128/dashboard/health # API health check curl http://localhost:20128/api/monitoring/health -```` +``` ### Runtime Storage -- Основно състояние: `${DATA_DIR}/storage.sqlite` (доставчици, комбинации, псевдоними, ключове, настройки) -- Използване: SQLite таблици в `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + незадължително `${DATA_DIR}/log.txt` и `${DATA_DIR}/call_logs/` -- Заявки за регистрационни файлове: `/logs/...` (като `ENABLE_REQUEST_LOGS=true`)---## Circuit Breaker Issues +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- + +## Circuit Breaker Issues ### Provider stuck in OPEN state -При прекъсване на веригата на доставчика е ОТВОРЕЕН, заявките се блокират, докато изтече времето за охлаждане. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Коригиране:** +**Fix:** -1. Отидете на**Табло → Настройки → Устойчивост** -2. Проверете картата на прекъсвача на сървъра на доставчика -3. Щракнете върху**Нулиране на всички**, за да изчистите всички прекъсвачи, или изчакайте времето за охлаждане да изтече -4. Уверете се, че доставчикът действително е наличен, преди да нулира### Доставчикът продължава да изключва прекъсвача +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Ако доставчикът многократно влезе в ОТВОРЕНО състояние: +### Provider keeps tripping the circuit breaker -1. Проверете**Таблото → Здраве → Здраве на доставчика**за модел на повреда -2. Отидете на**Настройки → Устойчивост → Профили на доставчици**и увеличите прага на отказ -3. Проверете дали доставчикът е променил ограниченията на API или изисква повторно удостоверяване -4. Прегледайте телеметрията за латентност — високата латентност може да причини грешки, базирани на изчакване---## Audio Transcription Issues +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- + +## Audio Transcription Issues ### "Unsupported model" error -- Уверете се, че използвате правилния префикс: `deepgram/nova-3` или `assemblyai/best` -- Проверете дали доставчикът е свързан в**Табло → Доставчици**### Транскрипцията се връща празна или е неуспешна +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Проверете поддържаните аудио формати: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- Уверете се, че размерът на файла е в границите на доставчика (обикновено < 25MB) -- Проверете валидността на API ключа на доставчика в картата на доставчика---## Translator Debugging +### Transcription returns empty or fails -Използвайте**Таблица за управление → Преводач**за отстраняване на грешки при проблеми с превод на формат: +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card -| Режим | Кога да използвате | -| ------------------- | ------------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Детска площадка** | Сравнете входно/изходните формати един до друг — поставете неуспешна заявка, за да видите как се превежда | -| **Чат тестер** | Изпращайте съобщения на живо и проверете допълнителен полезен продукт на заявка/отговор, включително заглавки | -| **Тестова стенда** | Изпълнете групови тестове в комбинации от формати, за да откриете кои преводи са нарушени | -| **Монитор на живо** | Гледайте потока на заявките в реално време, за да уловите периодични проблеми с превод | ### Common format issues | +--- --**Тагове за мислене не се появяват**— Проверете дали целевият доставчик поддържа мисленето и настройката на бюджета за мислене -**Отпадане на извикванията на инструментите**— Някои преводи на формати могат да премахнат неподдържаните полета; потвърдете в режим Playground -**Липсва системна подкана**— Клод и Джемини обработват системните подкани по различен начин; проверка на резултата за превод -**SDK връща необработен низ вместо валиден обект**— Коригирано във v1.1.0: дезинфектантът за отговор вече премахва нестандартните полета (`x_groq`, `usage_breakdown` и т.н.), които предизвикват неуспешно пускане на OpenAI SDK Pydantic -**GLM/ERNIE отхвърля `системна` роля**— Коригирано във v1.1.0: нормализаторът на ролите автоматично обединява системни съобщения в потребителски съобщения за несъвместими модели +## Translator Debugging -- Ролята на**`разработчик` не е разпозната**— Коригирано във v1.1.0: автоматично се преобразува в `системата` за доставчици, не е с OpenAI -**`json_schema` не работи с Gemini**— Коригирано във v1.1.0: `response_format` вече се преобразува в `responseMimeType` + `responseSchema` на Gemini---## Resilience Settings +Use **Dashboard → Translator** to debug format translation issues: + +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | + +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- + +## Resilience Settings ### Auto rate-limit not triggering -- Автоматичното ограничение на скоростта се прилага само за доставчици на API ключове (без OAuth/абонамент) -- Уверете се, че**Настройки → Устойчивост → Профили на доставчици**има активиран автоматичен лимит на скоростта -- Проверете дали доставчикът връща кодове за състояние `429` или заглавки `Retry-After`### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -Профилът на доставчика поддържа тези настройки: +### Tuning exponential backoff --**Базово плащане**— Първоначално време на изчакване след първото увреждане (по подразбиране: 1s) -**Максимално заплащане**— Максимално ограничение на времето за изчакване (по подразбиране: 30 секунди) -**Множител**— Колко да се увеличи закъснението за последователен отказ (по подразбиране: 2x)### Anti-thundering herd +Provider profiles support these settings: -Когато много едновременни заявки се появяват на доставчика с ограничена скорост, OmniRoute използва mutex + автоматично регулиране на скоростта, за да сериализира заявките и да предотврати каскадни грешки. Това е автоматично за доставчиците на API ключове.---## Optional RAG / LLM failure taxonomy (16 problems) +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) -Някои потребители на OmniRoute поставят шлюза пред RAG или агент стекове. В тези настройки е обичайно да се вижда отстранен модел: OmniRoute изглежда здрав (доставчиците работят, профилите за маршрутизиране са добри, няма предупреждения за ограничение на скоростта), но крайният отговор е още по-грешен. +### Anti-thundering herd -На практика тези инциденти идват от тръбопровода RAG надолу по веригата, а не от вашия шлюз. +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. -Ако търсите отделен речник, който да напише тези повреди, можете да използвате WFGY ProblemMap, външен текстов ресурс за лиценз на MIT, който дефинира шестнадесет повтарящи се модели при отказ на RAG / LLM. На високо ниво: обхваща +--- -- отклонение при извличане и нарушени контекстни граници -- празни или остарели индекси и векторни хранилища -- вграждане срещу семантично несъответствие -- бързо сглобяване и проблеми с контекстния прозорец -- логически колапс и изключително самоуверени отговори -- дълга верига и неуспехи в координацията на агента -- мултиагентна памет и дрейф на ролите -- проблеми с внедряването и подреждането на избраното зареждане +## Optional RAG / LLM failure taxonomy (16 problems) -Идеята е проста: +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -1. Когато проучвате лош отговор, заснемете: - - потребителска задача и заявка - - комбо маршрут или доставчик в OmniRoute - - всеки RAG контекст, използван надолу по веригата (извлечени документи, извиквания на инструменти и т.н.) -2. Съставете инцидента с едно или две номера на картата на проблемите на WFGY („No.1“ … „No.16“). -3. Съхранявайте номера във вашето собствено табло, runbook или инструмент за проследяване на инциденти до регистриране на файлове в OmniRoute. -4. Съществувате WFGY страница, за да решите дали трябва да промените своя RAG стек, ретривър или стратегия за маршрутизиране. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Пълният текст и конкретните рецепти се намират тук (лиценз на MIT, само текстът): +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: + +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems + +The idea is simple: + +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. + +Full text and concrete recipes live here (MIT license, text only): [WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -Можете да пренебрегнете този раздел, ако не изпълните RAG или конвейери на агенти зад OmniRoute.---## Still Stuck? +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. --**Проблеми с GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Архитектура**: Вижте [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) за вътрешни подробности -**API Reference**: Вижте [`docs/API_REFERENCE.md`](API_REFERENCE.md) за всички крайни точки -**Таблица за управление на здравето**: Проверете**Таблица за управление → Здраве**за състоянието на системата в реално време -**Преводач**: Използвайте**Табло за управление → Преводач**за отстраняване на грешки във форматирането +--- + +## Still Stuck? + +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/bg/llm.txt b/docs/i18n/bg/llm.txt new file mode 100644 index 0000000000..fe9a151253 --- /dev/null +++ b/docs/i18n/bg/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Български) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Преглед + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Сигурност +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/cs/README.md b/docs/i18n/cs/README.md index 7da2e6d4a3..6271fb5271 100644 --- a/docs/i18n/cs/README.md +++ b/docs/i18n/cs/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Váš univerzální API proxy – jeden koncový bod, 60+ poskytovatelů, nulové prostoje. Nyní s**MCP Server (25 nástrojů)**,**Protokol A2A**,**Paměť/Skills Systems**a**Electron Desktop App**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Dokončení chatu • Vložení • Generování obrázků • Video • Hudba • Zvuk • Změna pořadí •**Vyhledávání na webu**• Server MCP • Protokol A2A • 100% TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Váš univerzální API proxy – jeden koncový bod, 60+ poskytovatelů, nulov [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Web](https://omniroute.online) • [🚀 Rychlý start](#-rychlý start) • [💡 Funkce](#-klíčových-funkcí) • [📖 Dokumenty](#-dokumentace) • [💰 Cena](#-cena-na první pohled) • [🬒 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Dostupné v:**🇺🇸 [anglicky](README.md) | 🇧🇷 [Português (Brazílie)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dánsk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Maďarština](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugalsko)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,63 +60,65 @@ _Váš univerzální API proxy – jeden koncový bod, 60+ poskytovatelů, nulov ## 📸 Dashboard Preview - -Kliknutím zobrazíte snímky obrazovky řídicího panelu +
+Click to see dashboard screenshots -| Strana | Snímek obrazovky | -| --------------------- | --------------------------------------------------- | ---------- | -| **Poskytovatelé** | ![Poskytovatelé](docs/screenshots/01-providers.png) | -| **Komba** | ![Combos](docs/screenshots/02-combos.png) | -| **Analytika** | ![Analytics](docs/screenshots/03-analytics.png) | -| **Zdraví** | ![Zdraví](docs/screenshots/04-health.png) | -| **Překladatel** | ![Translator](docs/screenshots/05-translator.png) | -| **Nastavení** | ![Nastavení](docs/screenshots/06-settings.png) | -| **Nástroje CLI** | ![Nástroje CLI](docs/screenshots/07-cli-tools.png) | -| **Protokoly použití** | ![Použití](docs/screenshots/08-usage.png) | -| **Koncové body** | ![Koncové body](docs/screenshots/09-endpoint.png) |
| +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | + + --- ### 🤖 Free AI Provider for your favorite coding agents -_Připojte jakýkoli nástroj IDE nebo CLI s umělou inteligencí prostřednictvím OmniRoute – bezplatné brány API pro neomezené kódování._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ - + @@ -118,489 +127,562 @@ _Připojte jakýkoli nástroj IDE nebo CLI s umělou inteligencí prostřednictv OpenCode
OpenCode
- ⭐ 106 000 + ⭐ 106K
OpenClaw
OpenClaw

- ⭐ 205 000 + ⭐ 205K
NanoBot
NanoBot

- ⭐ 20,9 000 + ⭐ 20.9K
PicoClaw
PicoClaw

- ⭐ 14,6 000 + ⭐ 14.6K
ZeroClaw
ZeroClaw

- ⭐ 9,9 000 + ⭐ 9.9K
- Železný dráp
+ IronClaw
IronClaw

- ⭐ 2,1 000 + ⭐ 2.1K
Codex CLI
Codex CLI

- ⭐ 60,8 000 + ⭐ 60.8K
- Kód Claude
+ Claude Code
Claude Code

- ⭐ 67,3 000 + ⭐ 67.3K
Gemini CLI
Gemini CLI

- ⭐ 94,7 000 + ⭐ 94.7K
- Kilokód
- Kilový kód + Kilo Code
+ Kilo Code

- ⭐ 15,5 000 + ⭐ 15.5K
-📡 Všichni agenti se připojují přes http://localhost:20128/v1 nebo http://cloud.omniroute.online/v1 — jedna konfigurace, neomezené modely a kvóta--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Přestaňte plýtvat penězi a narážet na limity:** +**Stop wasting money and hitting limits:** -– Kvóta předplatného vyprší nevyužita každý měsíc -– Omezení sazby vám brání uprostřed kódování -– Drahá rozhraní API (20–50 $ měsíčně na poskytovatele) +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -- Ruční přepínání mezi poskytovateli +**OmniRoute solves this:** -**OmniRoute to řeší:** +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool -- ✅**Maximalizujte odběry**- Sledujte kvótu, před resetováním použijte každý bit -- ✅**Automatická záloha**- Předplatné → Klíč API → Levné → Zdarma, nulové prostoje -- ✅**Více účtů**- Round-robin mezi účty na poskytovatele -- ✅**Universal**- Funguje s Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, jakýmkoliv nástrojem CLI--- +--- ## 📧 Support -> 💬**Připojte se k naší komunitě!**[Skupina WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Získejte nápovědu, sdílejte tipy a buďte v obraze. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Web**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Problémy**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Skupina komunity](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Přispívání**: Podívejte se na [CONTRIBUTING.md](CONTRIBUTING.md), otevřete PR nebo si vyberte „dobré první číslo“ -**Původní projekt**: [9router by decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Při otevírání problému spusťte příkaz system-info a připojte vygenerovaný soubor:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Tím se vygeneruje soubor `system-info.txt` s vaší verzí Node.js, verzí OmniRoute, podrobnostmi OS, nainstalovanými nástroji CLI (qoder, gemini, claude, codex, antigravity, droid atd.), stavem Docker/PM2 a systémovými balíčky – vše, co potřebujeme k rychlé reprodukci vašeho problému. Připojte soubor přímo k vašemu problému na GitHubu.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Každý vývojář používající nástroje AI čelí těmto problémům denně.**OmniRoute byl vytvořen tak, aby je vyřešil všechny – od překročení nákladů po regionální bloky, od přerušených toků OAuth po operace protokolů a podniková pozorovatelnost. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. - -💸 1. „Platím za drahé předplatné, ale stále mě vyrušují limity“ +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Vývojáři platí 20–200 $ měsíčně za Claude Pro, Codex Pro nebo GitHub Copilot. I při placení má kvóta strop – 5 hodin používání, týdenní limity nebo limity sazby za minutu. Uprostřed relace kódování poskytovatel přestane reagovat a vývojář ztrácí tok a produktivitu. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Jak to řeší OmniRoute:** +**How OmniRoute solves it:** --**Chytrý 4-úrovňový záložní zdroj**– Pokud dojde k vyčerpání kvóty předplatného, automaticky se přesměruje na klíč API → Levné → Zdarma s nulovým ručním zásahem --**Sledování limitů poskytovatele**– Snímky kvót v mezipaměti se obnovují podle plánu na straně serveru (výchozí `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) s možností ručního obnovení v uživatelském rozhraní -–**Podpora více účtů**– Více účtů na poskytovatele s automatickým opakováním – když jeden dojde, přepne se na další --**Vlastní komba**– Přizpůsobitelné záložní řetězce s 9 strategiemi vyvažování (prioritní, vážená, na prvním místě, s cyklem, P2C, náhodná, nejméně používaná, nákladově optimalizovaná, striktně náhodná) --**Codex Business Quotas**— Sledování kvót Business/Tým pracovního prostoru přímo na řídicím panelu
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard - -🔌 2. „Potřebuji používat více poskytovatelů, ale každý má jiné API“ + -OpenAI používá jeden formát, Claude (Anthropic) jiný a Gemini ještě jiný. Pokud chce vývojář testovat modely od různých poskytovatelů nebo mezi nimi couvnout, musí překonfigurovat sady SDK, změnit koncové body, vypořádat se s nekompatibilními formáty. Vlastní poskytovatelé (FriendLI, NIM) mají nestandardní koncové body modelu. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Jak to řeší OmniRoute:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Unified Endpoint**– Jediný `http://localhost:20128/v1` slouží jako proxy pro všech 60+ poskytovatelů --**Formátový překlad**— Automatický a transparentní: OpenAI ↔ Claude ↔ Gemini ↔ Responses API --**Response Sanitization**– Odstraňuje nestandardní pole (`x_groq`, `usage_breakdown`, `service_tier`), která porušují OpenAI SDK v1.83+ --**Normalizace rolí**— Převádí `vývojář` → `systém` pro poskytovatele, kteří nejsou OpenAI; `systém` → `uživatel` pro GLM/ERNIE -–**Think Tag Extraction**– Extrahuje bloky „“ z modelů jako DeepSeek R1 do standardizovaného „reasoning_content“ --**Strukturovaný výstup pro Gemini**— `json_schema` → automatický převod `responseMimeType`/`responseSchema` --**`stream` má výchozí hodnotu `false`**— Vyhovuje specifikaci OpenAI a zabraňuje neočekávanému SSE v sadách Python/Rust/Go SDK
+**How OmniRoute solves it:** - -🌐 3. „Můj poskytovatel umělé inteligence blokuje můj region/země“ +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Poskytovatelé jako OpenAI/Codex blokují přístup z určitých geografických oblastí. Během připojení OAuth a rozhraní API se uživatelům zobrazují chyby jako „unsupported_country_region_territory“. To je frustrující zejména pro vývojáře z rozvojových zemí. + -**Jak to řeší OmniRoute:** +
+🌐 3. "My AI provider blocks my region/country" -–**3úrovňová konfigurace proxy**– konfigurovatelný proxy na 3 úrovních: globální (veškerý provoz), podle poskytovatele (pouze jeden poskytovatel) a podle připojení/klíče --**Barevně kódované odznaky proxy**— Vizuální indikátory: 🢢 globální proxy, 🟡 proxy poskytovatele, 🔵 proxy připojení, vždy zobrazující IP --**Výměna tokenů OAuth přes proxy**– tok OAuth prochází také přes proxy a řeší se `unsupported_country_region_territory` --**Testy připojení přes proxy**— Testy připojení používají nakonfigurovaný proxy (už žádné přímé obcházení) --**Podpora SOCKS5**— Plná podpora proxy SOCKS5 pro odchozí směrování --**TLS Fingerprint Spoofing**– TLS otisk prstu podobný prohlížeči přes `wreq-js` k obejití detekce botů --**🔏 CLI Fingerprint Matching**– Změní pořadí hlaviček a polí těla tak, aby odpovídaly nativním binárním podpisům CLI, čímž se výrazně sníží riziko označení účtu. IP proxy serveru je zachována – získáte současně maskování IP maskování**a**utajení
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. - -🆓 4. „Chci používat AI pro kódování, ale nemám peníze“ +**How OmniRoute solves it:** -Ne každý může platit 20–200 $ měsíčně za předplatné AI. Studenti, vývojáři z rozvíjejících se zemí, fandové a nezávislí pracovníci potřebují přístup ke kvalitním modelům za nulové náklady. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Jak to řeší OmniRoute:** + --**Vestavění poskytovatelé bezplatných úrovní**— Nativní podpora pro 100% bezplatné poskytovatele: Qoder (5 neomezených modelů prostřednictvím OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 neomezené modely: qwen3-qwender-lash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID zdarma), Gemini CLI (180 000 tokenů/měsíc zdarma) --**Ollama Cloud**– modely Ollama hostované v cloudu na `api.ollama.com` s bezplatnou úrovní „Light use“; použijte předponu `ollamacloud/` -–**komba pouze zdarma**– řetězec `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = 0 $/měsíc s nulovými prostoji --**Volný přístup NVIDIA NIM**— ~40 RPM pro vývojáře - navždy bezplatný přístup k více než 70 modelům na build.nvidia.com (přechod z kreditů na limity čisté sazby) --**Cost Optimized Strategy**– Strategie směrování, která automaticky vybírá nejlevnějšího dostupného poskytovatele +
+🆓 4. "I want to use AI for coding but I have no money" - -🔒 5. „Potřebuji chránit svou bránu AI před neoprávněným přístupem“ +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -Při vystavení brány AI do sítě (LAN, VPS, Docker) může kdokoli s adresou spotřebovat tokeny/kvótu vývojáře. Bez ochrany jsou rozhraní API zranitelná vůči zneužití, rychlému vložení a zneužití. +**How OmniRoute solves it:** -**Jak to řeší OmniRoute:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**Správa klíčů API**– Generování, rotace a rozsah podle poskytovatele pomocí vyhrazené stránky `/dashboard/api-manager` --**Oprávnění na úrovni modelu**– Omezte klíče API na konkrétní modely (`openai/*`, vzory zástupných znaků) pomocí přepínače Povolit vše/Omezit --**API Endpoint Protection**– Vyžadovat klíč pro `/v1/models` a blokovat konkrétní poskytovatele ze seznamu --**Auth Guard + ochrana CSRF**– Všechny cesty řídicího panelu chráněny middlewarem „withAuth“ + tokeny CSRF --**Rate Limiter**— omezení rychlosti na IP pomocí konfigurovatelných oken --**IP Filtering**— Seznam povolených/blokovaných pro řízení přístupu --**Prompt Injection Guard**– Dezinfekce proti škodlivým vzorům výzev --**Šifrování AES-256-GCM**— Přihlašovací údaje jsou v klidu zašifrovány
+ - -🛑 6. "Můj poskytovatel selhal a ztratil jsem tok kódování" +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -Poskytovatelé umělé inteligence se mohou stát nestabilními, vracet chyby 5xx nebo narazit na dočasné limity sazeb. Pokud vývojář závisí na jediném poskytovateli, je přerušen. Bez jističů mohou opakované pokusy způsobit selhání aplikace. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Jak to řeší OmniRoute:** +**How OmniRoute solves it:** --**Jistič pro každý model**— Automatické otevírání/zavírání s konfigurovatelnými prahovými hodnotami a ochlazením (zavřeno/otevřeno/polootevřeno), s rozsahem pro každý model, aby se zabránilo kaskádovým blokům --**Exponential Backoff**— Progresivní zpoždění opakování --**Anti-Thundering Herd**— Mutex + semaforová ochrana proti souběžným opakovaným bouřím --**Combo Fallback Chains**— Pokud primární poskytovatel selže, automaticky projde řetězcem bez zásahu --**Combo Circuit Breaker**– Automaticky deaktivuje selhávající poskytovatele v rámci kombinovaného řetězce -–**Health Dashboard**– Monitorování provozuschopnosti, stavy jističů, uzamčení, statistiky mezipaměti, latence p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest - -🔧 7. „Konfigurace každého nástroje umělé inteligence je únavná a opakující se“ + -Vývojáři používají Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Každý nástroj potřebuje jinou konfiguraci (API endpoint, klíč, model). Překonfigurování při změně poskytovatele nebo modelu je ztráta času. +
+🛑 6. "My provider went down and I lost my coding flow" -**Jak to řeší OmniRoute:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**CLI Tools Dashboard**– Vyhrazená stránka s nastavením jedním kliknutím pro Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline -–**GitHub Copilot Config Generator**– Generuje `chatLanguageModels.json` pro kód VS s hromadným výběrem modelu --**Průvodce přihlášením**– Průvodce nastavením ve 4 krocích pro začínající uživatele -–**Jeden koncový bod, všechny modely**– Jednou nakonfigurujte `http://localhost:20128/v1`, získáte přístup k více než 60 poskytovatelům
+**How OmniRoute solves it:** - -🔑 8. „Správa tokenů OAuth od více poskytovatelů je peklo“ +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot – všechny používají OAuth 2.0 s končícími tokeny. Vývojáři se musí neustále znovu autentizovat, řešit `client_secret is missing`, `redirect_uri_mismatch` a selhání na vzdálených serverech. Zvláště problematické je OAuth na LAN/VPS. + -**Jak to řeší OmniRoute:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Automatické obnovení tokenu**– Tokeny OAuth se před vypršením platnosti obnovují na pozadí --**Vestavěný OAuth 2.0 (PKCE)**— Automatický tok pro Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder -–**Multi-Account OAuth**– Více účtů na poskytovatele prostřednictvím extrakce tokenů JWT/ID --**OAuth LAN/Remote Fix**— Detekce privátní IP adresy pro `redirect_uri` + ruční režim URL pro vzdálené servery --**OAuth Behind Nginx**– Používá `window.location.origin` pro zpětnou kompatibilitu proxy -–**Průvodce vzdáleným OAuth**– Podrobný průvodce pro přihlašovací údaje Google Cloud na VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. - -📊 9. "Nevím, kolik a kde utrácím" +**How OmniRoute solves it:** -Vývojáři využívají více placených poskytovatelů, ale nemají jednotný pohled na výdaje. Každý poskytovatel má svůj vlastní panel fakturace, ale neexistuje žádné konsolidované zobrazení. Neočekávané náklady se mohou nahromadit. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Jak to řeší OmniRoute:** + --**Cost Analytics Dashboard**– Sledování nákladů na token a správa rozpočtu na poskytovatele --**Rozpočtové limity na úroveň**– Strop útraty na úroveň, který spouští automatickou rezervu --**Konfigurace cen za model**– Konfigurovatelné ceny za model --**Statistika využití na klíč API**— Počet požadavků a naposledy použité časové razítko na klíč -–**Panel Analytics**– Statistické karty, graf využití modelu, tabulka poskytovatelů s mírou úspěšnosti a latencí +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" - -🐛 10. „Nemohu diagnostikovat chyby a problémy ve voláních AI“ +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Když se volání nezdaří, vývojář neví, zda to byl limit sazby, vypršela platnost tokenu, nesprávný formát nebo chyba poskytovatele. Fragmentované protokoly napříč různými terminály. Bez pozorovatelnosti je ladění metodou pokus-omyl. +**How OmniRoute solves it:** -**Jak to řeší OmniRoute:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Sjednocený panel protokolů**– 4 karty: Protokoly požadavků, Protokoly proxy, Protokoly auditu, Konzole --**Console Log Viewer**— Prohlížeč ve stylu terminálu v reálném čase s barevně odlišenými úrovněmi, automatickým posouváním, vyhledáváním, filtrem --**Protokoly SQLite Proxy**— Trvalé protokoly, které vydrží restartování serveru --**Translator Playground**— 4 režimy ladění: Playground (překlad formátu), Tester chatu (zpáteční), Test Bench (dávka), Live Monitor (v reálném čase) -–**Požadavek na telemetrii**– latence p50/p95/p99 + sledování X-Request-Id --**Protokolování založené na souborech s rotací**– Protokoly aplikací se střídají podle velikosti, dnů uchování a počtu archivů; Artefakty protokolu hovorů rotují podle dnů uchování a počtu souborů --**System Info Report**— `npm run system-info` vygeneruje `system-info.txt` s vaším úplným prostředím (verze uzlu, verze OmniRoute, OS, nástroje CLI, stav Docker/PM2). Připojte jej při hlášení problémů pro okamžité třídění.
+ - -🏗️ 11. „Nasazení a údržba brány je složitá“ +
+📊 9. "I don't know how much I'm spending or where" -Instalace, konfigurace a údržba AI proxy v různých prostředích (místní, VPS, Docker, cloud) je náročná na práci. Problémy jako pevně zakódované cesty, „EACCES“ v adresářích, konflikty portů a sestavení napříč platformami zvyšují tření. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Jak to řeší OmniRoute:** +**How OmniRoute solves it:** --**Globální instalace npm**– `npm install -g omniroute && omniroute` – hotovo --**Docker Multi-Platform**– nativní AMD64 + ARM64 (Apple Silicon, AWS Graviton, Raspberry Pi) --**Docker Compose Profiles**– `base` (žádné nástroje CLI) a `cli` (s Claude Code, Codex, OpenClaw) --**Electron Desktop App**– nativní aplikace pro Windows/macOS/Linux se systémovou lištou, automatickým spuštěním, offline režimem --**Split-Port Mode**– API a Dashboard na samostatných portech pro pokročilé scénáře (reverzní proxy, kontejnerová síť) --**Cloud Sync**— Konfigurace synchronizace mezi zařízeními pomocí Cloudflare Workers --**DB Backups**— Automatické zálohování, obnova, export a import všech nastavení s `DISABLE_SQLITE_AUTO_BACKUP` pro externě spravované zálohy
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency - -🌍 12. "Rozhraní je pouze v angličtině a můj tým nemluví anglicky" + -Týmy v neanglicky mluvících zemích, zejména v Latinské Americe, Asii a Evropě, se potýkají s rozhraním pouze v angličtině. Jazykové bariéry snižují přijetí a zvyšují chyby konfigurace. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Jak to řeší OmniRoute:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Dashboard i18n — 30 jazyků**— Všech 500+ kláves přeloženo včetně arabštiny, bulharštiny, dánštiny, němčiny, španělštiny, finštiny, francouzštiny, hebrejštiny, hindštiny, maďarštiny, indonéštiny, italštiny, japonštiny, korejštiny, malajštiny, holandštiny, norštiny, polštiny, portugalštiny (PT/BR), rumunštiny, ruštiny, slovenštiny, švédštiny, thajštiny, filipínštiny, vietnamštiny, angličtiny --**Podpora RTL**— Podpora zprava doleva pro arabštinu a hebrejštinu --**Vícejazyčné README**— 30 kompletních překladů dokumentace --**Language Selector**— Ikona zeměkoule v záhlaví pro přepínání v reálném čase
+**How OmniRoute solves it:** - -🔄 13. „Potřebuji víc než jen chat – potřebuji vložení, obrázky, zvuk“ +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -AI není jen dokončení chatu. Vývojáři potřebují generovat obrázky, přepisovat zvuk, vytvářet vložení pro RAG, měnit hodnocení dokumentů a moderovat obsah. Každé API má jiný koncový bod a formát. + -**Jak to řeší OmniRoute:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Vložení**— `/v1/embeddings` se 6 poskytovateli a 9+ modely --**Generace obrázků**— `/v1/images/generations` s 10 poskytovateli a 20+ modely (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Text-to-Video**— `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) a SD WebUI --**Text-to-Music**— `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) --**Audio Transscription**— `/v1/audio/transscriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Text-to-Speech**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + stávající poskytovatelé --**Moderations**— `/v1/moderations` — Kontroly bezpečnosti obsahu --**Přehodnocení**— `/v1/rerank` — Změna pořadí podle relevance dokumentu --**Responses API**— Plná podpora `/v1/responses` pro Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. - -🧪 14. „Nemám způsob, jak testovat a porovnávat kvalitu napříč modely“ +**How OmniRoute solves it:** -Vývojáři chtějí vědět, který model je pro jejich případ použití nejlepší – kód, překlad, uvažování – ale ruční porovnávání je pomalé. Neexistují žádné integrované nástroje eval. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**Jak to řeší OmniRoute:** + --**Hodnocení LLM**– Testování zlaté sady s 10 předem nahranými případy zahrnujícími pozdravy, matematiku, geografii, generování kódu, soulad s JSON, překlad, markdown, bezpečnostní odmítnutí --**4 strategie shody**– `přesné`, `obsahuje`, `regulární výraz`, `vlastní` (funkce JS) --**Testovací stolice pro překladatelské hřiště**– Dávkové testování s více vstupy a očekávanými výstupy, porovnání mezi poskytovateli --**Chat Tester**– Kompletní zpáteční cesta s vykreslováním vizuální odezvy --**Live Monitor**— Tok všech požadavků procházejících přes proxy v reálném čase +
+🌍 12. "The interface is English-only and my team doesn't speak English" - -📈 15. „Potřebuji škálovat bez ztráty výkonu“ +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -Jak roste objem požadavků, bez ukládání stejných otázek do mezipaměti vznikají duplicitní náklady. Bez idempotence duplikát požaduje zpracování odpadu. Musí být dodrženy limity sazeb na poskytovatele. +**How OmniRoute solves it:** -**Jak to řeší OmniRoute:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Sémantická mezipaměť**– Dvouvrstvá mezipaměť (podpis + sémantická) snižuje náklady a latenci --**Idempotency požadavku**— 5s deduplikační okno pro identické požadavky -–**Detekce limitu rychlosti**– RPM na poskytovatele, minimální mezera a maximální souběžné sledování --**Upravitelné limity rychlosti**– Konfigurovatelné výchozí hodnoty v Nastavení → Odolnost s perzistencí --**API Key Validation Cache**– 3vrstvá mezipaměť pro produkční výkon -–**Health Dashboard s telemetrií**– latence p50/p95/p99, statistiky mezipaměti, doba provozu
+ - -🤖 16. „Chci globálně ovládat chování modelu“ +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Vývojáři, kteří chtějí všechny odpovědi v konkrétním jazyce, s konkrétním tónem nebo chtějí omezit tokeny uvažování. Konfigurace tohoto v každém nástroji/požadavku je nepraktická. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Jak to řeší OmniRoute:** +**How OmniRoute solves it:** --**System Prompt Injection**– Globální výzva aplikovaná na všechny požadavky --**Thinking Budget Validation**— Řízení alokace tokenů na základě požadavku (průchozí, automatické, vlastní, adaptivní) --**9 směrovacích strategií**— Globální strategie, které určují způsob distribuce požadavků --**Wildcard Router**– vzory `poskytovatel/*` se dynamicky směrují k libovolnému poskytovateli --**Povolit/zakázat přepínání komba**— Přepínejte komba přímo z řídicího panelu --**Přepnutí poskytovatele**— Povolí/zakáže všechna připojení pro poskytovatele jedním kliknutím -–**Blokovaní poskytovatelé**– vyloučení konkrétních poskytovatelů ze seznamu `/v1/models`
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex - -🧰 17. „Potřebuji nástroje MCP jako prvotřídní možnosti produktu“ + -Mnoho bran AI odhaluje MCP pouze jako skrytý detail implementace. Týmy potřebují viditelnou a spravovatelnou provozní vrstvu. +
+🧪 14. "I have no way to test and compare quality across models" -**Jak to řeší OmniRoute:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP se objeví na navigačním panelu a na kartě protokolu koncového bodu -- Vyhrazená stránka správy MCP s procesem, nástroji, rozsahy a auditem -- Vestavěný rychlý start pro `omniroute --mcp` a přihlášení klienta
+**How OmniRoute solves it:** - -🧠 18. „Potřebuji orchestraci A2A s cestami synchronizace + streamování“ +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Pracovní postupy agentů vyžadují jak přímé odpovědi, tak dlouhotrvající streamované spouštění s řízením životního cyklu. + -**Jak to řeší OmniRoute:** +
+📈 15. "I need to scale without losing performance" -– Koncový bod A2A JSON-RPC (`POST /a2a`) s `zprávou/odeslat` a `zprávou/streamem` -- SSE streamování s šířením koncového stavu -- Rozhraní API životního cyklu úloh pro `tasks/get` a `tasks/cancel`
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. - -🛰️ 19. „Potřebuji skutečný stav procesu MCP, nikoli odhadovaný stav“ +**How OmniRoute solves it:** -Operační týmy potřebují vědět, zda je MCP skutečně naživu, nejen zda je API dosažitelné. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**Jak to řeší OmniRoute:** + -- Soubor srdečního tepu za běhu s PID, časovými razítky, transportem, počtem nástrojů a režimem rozsahu -- Stavové API MCP kombinující srdeční tep + nedávnou aktivitu -- Stavové karty uživatelského rozhraní pro aktuálnost procesu / provozuschopnosti / srdečního tepu +
+🤖 16. "I want to control model behavior globally" - -📋 20. „Potřebuji provádění auditovatelného nástroje MCP“ +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Když nástroje mutují konfiguraci nebo spouštějí operace operací, týmy potřebují forenzní sledovatelnost. +**How OmniRoute solves it:** -**Jak to řeší OmniRoute:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- Protokolování auditu podporované SQLite pro volání nástrojů MCP -- Filtry podle nástroje, úspěchu/neúspěchu, klíče API a stránkování -- Tabulka auditu řídicího panelu + statistiky koncových bodů pro automatizaci
+ - -🔐 21. „Potřebuji omezená oprávnění MCP na integraci“ +
+🧰 17. "I need MCP tools as first-class product capabilities" -Různí klienti by měli mít nejméně privilegovaný přístup ke kategoriím nástrojů. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Jak to řeší OmniRoute:** +**How OmniRoute solves it:** -- 10 granulárních rozsahů MCP pro řízený přístup k nástrojům -- Vynucení rozsahu a viditelnost v uživatelském rozhraní správy MCP -- Bezpečná výchozí poloha pro provozní nástroje
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding - -⚙️ 22. „Potřebuji provozní kontroly bez přerozdělování“ + -Týmy potřebují rychlé změny běhového prostředí během incidentů nebo nákladových událostí. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**Jak to řeší OmniRoute:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Aktivace kombinace přepínačů přímo z řídicího panelu MCP -- Použijte profily odolnosti z předdefinovaných balíčků zásad -- Resetujte stav jističe ze stejného ovládacího panelu
+**How OmniRoute solves it:** - -🔄 23. „Potřebuji živou viditelnost a zrušení životního cyklu úkolu A2A“ +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Bez viditelnosti životního cyklu je obtížné třídit incidenty úkolů. + -**Jak to řeší OmniRoute:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Seznam úkolů / filtrování podle stavu / dovedností se stránkováním -- Podrobnější informace o metadatech úkolů, událostech a artefaktech -- Koncový bod zrušení úlohy a akce uživatelského rozhraní s potvrzením
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. - -🌊 24. „Potřebuji aktivní metriky streamu pro zatížení A2A“ +**How OmniRoute solves it:** -Streamovací pracovní postupy vyžadují provozní přehled o souběžných a živých připojeních. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**Jak to řeší OmniRoute:** + -- Aktivní čítače toku integrované do stavu A2A -- Časové razítko posledního úkolu a počty za stav -- Karty palubní desky A2A pro monitorování operací v reálném čase +
+📋 20. "I need auditable MCP tool execution" - -🪪 25. „Potřebuji pro klienty zjišťování standardních agentů“ +When tools mutate config or trigger ops actions, teams need forensic traceability. -Externí klienti a orchestrátoři potřebují strojově čitelná metadata pro integraci. +**How OmniRoute solves it:** -**Jak to řeší OmniRoute:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- Karta agenta vystavena na adrese `/.well-known/agent.json` -- Schopnosti a dovednosti zobrazené v uživatelském rozhraní pro správu -- A2A status API obsahuje metadata zjišťování pro automatizaci
+ - -🧭 26. „Potřebuji zjistitelnost protokolu v uživatelském rozhraní produktu“ +
+🔐 21. "I need scoped MCP permissions per integration" -Pokud uživatelé nemohou objevit protokolové povrchy, kvalita přijetí a podpory klesá. +Different clients should have least-privilege access to tool categories. -**Jak to řeší OmniRoute:** +**How OmniRoute solves it:** -– Konsolidovaná stránka**Koncové body**s kartami pro koncové body proxy, MCP, A2A a API -- Přepínání stavu inline služby (Online/Offline) pro MCP a A2A -- Odkazy z přehledu na vyhrazené karty správy
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling - -🧪 27. „Potřebuji komplexní ověření protokolu se skutečnými klienty“ + -Falešné testy nestačí k ověření kompatibility protokolu před vydáním. +
+⚙️ 22. "I need operational controls without redeploying" -**Jak to řeší OmniRoute:** +Teams need quick runtime changes during incidents or cost events. -- Sada E2E, která spouští aplikaci a používá skutečný přenos klienta MCP SDK -- Klient A2A testuje toky zjišťování, odesílání, streamování, získávání a rušení -- Křížová kontrola tvrzení proti auditu MCP a API úloh A2A
+**How OmniRoute solves it:** - -📡 28. „Potřebuji jednotnou pozorovatelnost napříč všemi rozhraními“ +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -Rozdělení pozorovatelnosti protokolem vytváří slepá místa a delší MTTR. + -**Jak to řeší OmniRoute:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Sjednocené dashboardy/logy/analýzy v jednom produktu -- Zdraví + audit + telemetrie požadavků napříč vrstvami OpenAI, MCP a A2A -- Provozní API pro stav a automatizaci
+Without lifecycle visibility, task incidents become hard to triage. - -💼 29. "Potřebuji jeden runtime pro proxy + nástroje + orchestraci agenta" +**How OmniRoute solves it:** -Provozování mnoha samostatných služeb zvyšuje provozní náklady a způsoby selhání. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**Jak to řeší OmniRoute:** + -- Proxy, MCP server a A2A server v jednom zásobníku kompatibilní s OpenAI -- Sdílená autentizace, odolnost, úložiště dat a pozorovatelnost -- Konzistentní model politiky na všech interakčních plochách +
+🌊 24. "I need active stream metrics for A2A load" - -🚀 30. „Potřebuji odeslat agentské pracovní postupy bez rozšiřování kódu lepidla“ +Streaming workflows require operational insight into concurrency and live connections. -Týmy ztrácejí rychlost při spojování více ad-hoc služeb a skriptů. +**How OmniRoute solves it:** -**Jak to řeší OmniRoute:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Jednotná strategie koncových bodů pro klienty a agenty -- Vestavěná uživatelská rozhraní pro správu protokolů a cesty ověřování kouře -- Základy připravené na výrobu (zabezpečení, protokolování, odolnost, zálohování)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Příručka A: Maximalizujte placené předplatné + levné zálohování**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -608,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Příručka B: Sada kódování s nulovými náklady**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Příručka C: 24/7 vždy zapnutý záložní řetězec**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -631,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Příručka D: Operace agenta s MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Nastavení kódování AI během několika minut za**$0/měsíc**. Propojte tyto bezplatné účty a použijte vestavěnou kombinaci**Free Stack**. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Krok | Akce | Poskytovatelé odemčeni | -| ---- | --------------------------------------------------- | ------------------------------------------------------------------- | -| 1 | Connect**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**neomezeno**| -| 2 | Připojte**Qoder**(Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... —**bez omezení**| -| 3 | Připojte**Qwen**(kód zařízení) | qwen3-coder-plus, qwen3-coder-flash... —**bez omezení**| -| 4 | Připojte**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180 000/měsíc zdarma**| -| 5 | `/dashboard/combos` → Šablona**Free Stack (0 $)**| Round-robin všechny bezplatné poskytovatele automaticky | +| Step | Action | Providers Unlocked | +| ---- | -------------------------------------------------- | ------------------------------------------------------------------ | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Nasměrujte libovolné IDE/CLI na:**`http://localhost:20128/v1` · Klíč API: `any-string` · Hotovo. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Volitelné dodatečné pokrytí (také zdarma):**Klíč Groq API (30 RPM zdarma), NVIDIA NIM (40 RPM zdarma, 70+ modelů), Cerebras (1 M token/den), LongCat API klíč (50 M tokenů/den!), Cloudflare Workers AI (10 000 neuronů/den, 50+ modelů).## Rychlý start +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Rychlý start ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **Uživatelé pnpm:**Po instalaci spusťte příkaz `pnpm accept-builds -g`, abyste povolili nativní skripty sestavení vyžadované `better-sqlite3` a `@swc/core`: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > > ```bash > pnpm install -g omniroute -> pnpm schválit-builds -g # Vybrat všechny balíčky → schválit +> pnpm approve-builds -g # Select all packages → approve > omniroute > ``` -Dashboard se otevře na adrese `http://localhost:20128` a základní adresa URL API je `http://localhost:20128/v1`. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Příkaz | Popis | -| ----------------------- | ------------------------------------------------------------------ | -| "všestranná cesta" | Spustit server (`PORT=20128`, API a řídicí panel na stejném portu) | -| `omniroute --port 3000` | Nastavte kanonický/API port na 3000 | -| `omniroute --mcp` | Spustit MCP server (stdio transport) | -| `omniroute --no-open` | Neotevírat automaticky prohlížeč | -| `omniroute --help` | Zobrazit nápovědu | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Volitelný režim rozděleného portu:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -Pro většinu nasazení potřebujete pouze: +For most deployments, you only need: -| Proměnná | Výchozí | Účel | -| ------------------------- | ------------------------------ | ------------------------------------------------------------- -------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | "600 000" | Sdílená základní linie pro upstream načítání, skryté časové limity Undici, požadavky otisků prstů TLS a časové limity požadavků/proxy mostu API | -| `STREAM_IDLE_TIMEOUT_MS` | zdědí `REQUEST_TIMEOUT_MS` | Maximální mezera mezi streamovanými bloky, než OmniRoute přeruší stream SSE | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -Zpětná kompatibilita je zachována: stávající `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` a další proměnná časového limitu pro jednotlivé vrstvy stále fungují a přepisují sdílenou základní linii. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Pokud potřebujete jemnější ovládání, jsou k dispozici pokročilé přepisy:| Proměnná | Výchozí | Účel | -| ----------------------------------------- | ------------------------------------------- | --------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | zdědí `REQUEST_TIMEOUT_MS` | Celkový časový limit upstream požadavku použitý signálem přerušení hlavního načítání | -| `FETCH_HEADERS_TIMEOUT_MS` | zdědí `FETCH_TIMEOUT_MS` | Undici časový limit pro příjem upstream hlaviček odpovědí | -| `FETCH_BODY_TIMEOUT_MS` | zdědí `FETCH_TIMEOUT_MS` | Undici časový limit mezi upstream body těla (`0` to zakáže) | -| `FETCH_CONNECT_TIMEOUT_MS` | "30 000" | Vypršel časový limit připojení Undici TCP | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | "4000" | Undici nečinný keep-alive socket timeout | -| `TLS_CLIENT_TIMEOUT_MS` | zdědí `FETCH_TIMEOUT_MS` | Vypršel časový limit pro požadavky otisku TLS provedené prostřednictvím `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | zdědí `REQUEST_TIMEOUT_MS` nebo `30000` | Časový limit pro přesměrování proxy `/v1` z portu API na port řídicího panelu | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Časový limit příchozího požadavku na serveru API mostu | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | "60 000" | Časový limit příchozí hlavičky na serveru API mostu | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | "5000" | Udržovací časový limit na serveru API mostu | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | "0" | Časový limit nečinnosti soketu na serveru API mostu (`0` jej zakáže) | +Advanced overrides are available if you need finer control: -Pokud spouštíte OmniRoute za Nginx, Caddy, Cloudflare nebo jiným reverzním proxy, ujistěte se, že proxy -časové limity jsou také vyšší než časové limity streamu/načtení OmniRoute.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Otevřete Dashboard → `Providers` a připojte alespoň jednoho poskytovatele (OAuth nebo API klíč). -2. Otevřete Dashboard → `Koncové body` a vytvořte klíč API. -3. (Volitelné) Otevřete Dashboard → `Komba` a nastavte svůj záložní řetězec.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Pracuje s Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode a SDK kompatibilní s OpenAI.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (pro operace řízené nástrojem):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -Poté připojte svého MCP klienta přes `stdio` a otestujte nástroje jako: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` -- `combos_omniroute_list_combos` +- `omniroute_list_combos` -**A2A (pro pracovní postupy mezi agenty):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -760,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Tato sada ověřuje skutečné klientské toky MCP a A2A proti běžící aplikaci.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -768,13 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` - -Void Linux (šablona `xbps-src`) +
+Void Linux (`xbps-src` template) -Pro uživatele Void Linuxu můžete vytvořit nativní balíček pomocí `xbps-src`. Uložte tento blok jako `srcpkgs/omniroute/template`:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -786,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -794,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -870,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -881,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute je k dispozici jako veřejný obrázek Dockeru na [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Rychlý běh:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -891,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**Se souborem prostředí:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Použití Docker Compose:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Podpora řídicího panelu pro nasazení Dockeru nyní zahrnuje**Cloudflare Quick Tunnel**na jedno kliknutí na `Dashboard → Endpoints`. První povolí stahování `cloudflared` pouze v případě potřeby, spustí dočasný tunel k vašemu aktuálnímu koncovému bodu `/v1` a zobrazí vygenerovanou URL `https://*.trycloudflare.com/v1` přímo pod vaší normální veřejnou adresou URL. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Poznámky: +Notes: -- URL Quick Tunnel jsou dočasné a mění se po každém restartu. -- Rychlé tunely se po restartu OmniRoute nebo kontejneru automaticky neobnoví. V případě potřeby je znovu povolte z řídicího panelu. -- Spravovaná instalace aktuálně podporuje Linux, macOS a Windows na `x64` / `arm64`. -- Spravované rychlé tunely jsou výchozí pro přenos HTTP/2, aby se zabránilo hlučným varováním vyrovnávací paměti QUIC UDP v prostředí s omezenými kontejnery. Pokud chcete jiný přenos, nastavte `CLOUDFLARED_PROTOCOL=quic` nebo `auto`. -- Obrazy Dockeru svazují kořeny systémové CA a předávají je spravovanému `cloudflared`, což zabraňuje selhání důvěryhodnosti TLS při zavádění tunelu uvnitř kontejneru. -- SQLite běží v režimu WAL. `Docker stop` by mělo být povoleno dokončit, aby OmniRoute mohl zkontrolovat nejnovější změny zpět do `storage.sqlite`. -- V přiložených souborech Compose je již nastavena doba odkladu 40 s. Pokud spouštíte obraz přímo, ponechte hodnotu `--stop-timeout 40` (nebo podobnou), aby ruční zastavení nepřerušilo čištění při vypnutí. -- Nastavte `CLOUDFLARED_BIN=/absolutní/cesta/k/cloudflared`, pokud chcete, aby OmniRoute místo stahování používal existující binární soubor. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Použití Docker Compose s Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute lze bezpečně zpřístupnit pomocí automatického zřizování SSL Caddy. Ujistěte se, že záznam DNS A vaší domény ukazuje na IP adresu vašeho serveru.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` +| Image | Tag | Size | Description | +| ------------------------ | -------- | ------ | --------------------- | +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | -| Obrázek | Štítek | Velikost | Popis | -| ------------------------- | -------- | ------ | ---------------------- | -| `diegosouzapw/omniroute` | "nejnovější" | ~250 MB | Poslední stabilní verze | -| `diegosouzapw/omniroute` | "1.0.3" | ~250 MB | Aktuální verze |--- +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**NOVINKA!**OmniRoute je nyní k dispozici jako**nativní desktopová aplikace**pro Windows, macOS a Linux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Spusťte OmniRoute jako samostatnou desktopovou aplikaci – pro místní modely není potřeba žádný terminál, žádný prohlížeč ani internet. Aplikace založená na Electronu zahrnuje: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Nativní okno**— Vyhrazené okno aplikace s integrací na systémové liště -- 🔄**Auto-Start**– Spusťte OmniRoute při přihlášení do systému -- 🔔**Nativní oznámení**– Získejte upozornění na vyčerpání kvóty nebo problémy s poskytovatelem -- ⚡**Instalace jedním kliknutím**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Režim offline**– Funguje plně offline s přibaleným serverem### Rychlý start +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Rychlý start ```bash # Development mode @@ -980,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -Když je minimalizován, OmniRoute žije v systémové liště s rychlými akcemi: +When minimized, OmniRoute lives in your system tray with quick actions: -- Otevřete palubní desku -- Změňte port serveru -- Ukončete aplikaci +- Open dashboard +- Change server port +- Quit application -📖 Úplná dokumentace: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Úroveň | Poskytovatel | Cena | Obnovení kvóty | Nejlepší pro | -| ----------------- | --------------------------- | ------------------------------- | --------------------------- | ------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 PŘEDPLATNÉ** | Claude Code (Pro) | 20 $/měsíc | 5h + týdně | Již přihlášeno | -| | Codex (Plus/Pro) | 20–200 USD/měsíc | 5h + týdně | Uživatelé OpenAI | -| | Gemini CLI | **ZDARMA** | 180 tis./měsíc + 1 tis./den | Každý! | -| | GitHub Copilot | 10–19 USD/měsíc | Měsíčně | Uživatelé GitHubu | -| **🔑 API KEY** | NVIDIA NIM | **ZDARMA**(dev forever) | ~40 RPM | 70+ otevřených modelů | -| | Cerebras | **ZDARMA**(1 milion toku/den) | 60 000 TPM / 30 RPM | Nejrychlejší na světě | -| | Groq | **ZDARMA**(30 RPM) | 14,4K RPD | Ultra rychlá lama/gemma | -| | DeepSeek V3.2 | 0,27 $ / 1,10 $ za 1 milion | Žádné | Nejlepší zdůvodnění cena/kvalita | -| | xAI Grok-4 Fast | **0,20 $/0,50 $ za 1M**🆕 | Žádné | Nejrychlejší + volání nástroje, ultranízké | -| | xAI Grok-4 (standardní) | 0,20 $/1,50 $ za 1M 🆕 | Žádné | Reasoning vlajková loď od xAI | -| | Mistral | Vyzkoušení zdarma + placené | Omezená sazba | Evropská umělá inteligence | -| | OpenRouter | Platba za použití | Žádné | 100+ modelů agr. | -| **💰 LEVNĚ** | GLM-5 (přes Z.AI) 🆕 | 0,5 $/1 mil. | Denně 10:00 | 128K výstup, nejnovější vlajková loď | -| | GLM-4.7 | 0,6 $/1 mil. | Denně 10:00 | Záloha rozpočtu | -| | MiniMax M2,5 🆕 | Vstup 0,3 $/1 milion | 5hodinové válcování | Úvahy + agentské úkoly | -| | MiniMax M2.1 | 0,2 $/1 milion | 5hodinové válcování | Nejlevnější varianta | -| | Kimi K2.5 (Moonshot API) 🆕 | Platba za použití | Žádné | Přímý přístup Moonshot API | -| | Kimi K2 | 9 $/měsíc byt | 10 milionů tokenů/měsíc | Předvídatelné náklady | -| **🆓 ZDARMA** | Qoder | **$0** | Neomezené | 5 modelů neomezeně | -| | Qwen | **$0** | Neomezené | 4 modely neomezeně | -| | Kiro | **$0** | Neomezené | Claude Sonnet/Haiku (stavitel AWS) | -| | LongCat Flash-Lite 🆕 | **$0**(50 milionů toku/den 🔥) | 1 RPS | Největší bezplatná kvóta na Zemi | -| | Opylování AI 🆕 | **$0**(není potřeba žádný klíč) | 1 požadavek/15s | GPT-5, Claude, DeepSeek, Llama 4 | -| | Cloudflare Workers AI 🆕 | **$0**(10 000 neuronů/den) | ~150 resp./den | 50+ modelů, globální náskok | -| | Scaleway AI 🆕 | **$0**(celkem 1 milion tokenů) | Omezená sazba | EU/GDPR, Qwen3 235B, Lama 70B | > 🆕**Přidané nové modely (březen 2026):**Rodina Grok-4 Fast za 0,20 $/0,50 $/M (porovnávací rychlost 1143 ms – o 30 % rychlejší než Gemini 2.5 Flash), GLM-5 přes Z.AI s výstupem 128K, aktualizovaná cena MiniMax M2.5 V5, přímé zdůvodnění Kimi K2.2 Moon. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 Combo Stack 0 $ — Kompletní bezplatné nastavení:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Nulové náklady. Nikdy nepřestane kódovat.**Nakonfigurujte si to jako jednu kombinaci OmniRoute a všechna nouzová řešení se stanou automaticky – žádné ruční přepínání.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Všechny níže uvedené modely jsou**100% zdarma bez nutnosti použití kreditní karty**. OmniRoute mezi nimi automaticky směruje, když dojde jedna kvóta – zkombinujte je všechny a získáte nerozbitnou kombinaci 0 $.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Model | Předpona | Limit | Limit sazby | -| -------------------- | ------ | ------------- | ---------------------- | -| `claude-sonnet-4.5` | `kr/` |**Neomezeno**| Žádný hlášený denní limit | -| `claude-haiku-4,5` | `kr/` |**Neomezeno**| Žádný hlášený denní limit | -| `claude-opus-4.6` | `kr/` |**Neomezeno**| Nejnovější Opus přes Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) -| Model | Předpona | Limit | Limit sazby | -| ------------------- | ------ | ------------- | ---------------- | -| "kimi-k2-myšlení" | `jestli/` |**Neomezeno**| Žádný nahlášený strop | -| `qwen3-coder-plus` | `jestli/` |**Neomezeno**| Žádný nahlášený strop | -| `deepseek-r1` | `jestli/` |**Neomezeno**| Žádný nahlášený strop | -| `minimax-m2.1` | `jestli/` |**Neomezeno**| Žádný nahlášený strop | -| "kimi-k2" | `jestli/` |**Neomezeno**| Žádný nahlášený strop | +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | --------------------- | +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -> Doporučený způsob připojení:**Personal Access Token + `qodercli`**. OAuth prohlížeče je -> experimentální a ve výchozím nastavení zakázáno, pokud nejsou nakonfigurovány proměnné prostředí `QODER_OAUTH_*`.### 🟡 QWEN MODELS (Device Code Auth) +### 🟢 QODER MODELS (Free PAT via qodercli) -| Model | Předpona | Limit | Limit sazby | -| -------------------- | ------ | ------------- | -------------------- | -| `qwen3-coder-plus` | `qw/` |**Neomezeno**| Žádný nahlášený strop | -| `qwen3-coder-flash` | `qw/` |**Neomezeno**| Žádný nahlášený strop | -| `qwen3-coder-next` | `qw/` |**Neomezeno**| Žádný nahlášený strop | -| "model vidění" | `qw/` |**Neomezeno**| Multimodální (obrázky) |### 🟣 GEMINI CLI (Google OAuth) +| Model | Prefix | Limit | Rate Limit | +| ------------------ | ------ | ------------- | --------------- | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -| Model | Předpona | Limit | Limit sazby | -| ------------------------- | ------ | ---------------------------- | ------------- | -| `gemini-3-flash-preview` | `gc/` |**180 tis./měsíc**+ 1 tis./den | Měsíční reset | -| `gemini-2.5-pro` | `gc/` | 180 tis./měsíc (sdílený bazén) | Vysoká kvalita |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| Úroveň | Denní limit | Limit sazby | Poznámky | -| ---------- | ------------ | ----------- | ------------------------------------------------------- | -| Zdarma (Dev) | Žádný token cap |**~40 RPM**| 70+ modelů; přechod na limity čisté sazby v polovině roku 2025 | +### 🟡 QWEN MODELS (Device Code Auth) -Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | ------------------- | +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Úroveň | Denní limit | Limit sazby | Poznámky | -| ---- | ------------------ | ----------------- | -------------------------------------------- | -| Zdarma |**1 mil. tokenů/den**| 60 000 TPM / 30 RPM | Světově nejrychlejší odvození LLM; resetuje denně | +### 🟣 GEMINI CLI (Google OAuth) -Dostupné zdarma: `lama-3.3-70b`, `lama-3.1-8b`, `deepseek-r1-distill-lama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -| Úroveň | Denní limit | Limit sazby | Poznámky | -| ---- | ------------- | ----------------- | ------------------------------------------ | -| Zdarma |**14,4K RPD**| 30 ot./min na model | Žádná kreditní karta; 429 na limit, neúčtuje se | +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) -Dostupné zdarma: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +| Tier | Daily Limit | Rate Limit | Notes | +| ---------- | ------------ | ----------- | ------------------------------------------------------ | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -| Model | Předpona | Denní kvóta zdarma | Poznámky | -| ------------------------------ | ------ | ------------------ | ------------------------ | -| "LongCat-Flash-Lite" | `lc/` |**50 milionů tokenů**💥 | Největší bezplatná kvóta všech dob | -| "LongCat-Flash-Chat" | `lc/` | 500 000 tokenů | Víceotáčkový chat | -| "LongCat-Flash-Thinking" | `lc/` | 500 000 tokenů | Zdůvodnění / CoT | -| "LongCat-Flash-Thinking-2601" | `lc/` | 500 000 tokenů | Verze z ledna 2026 | -| "LongCat-Flash-Omni-2603" | `lc/` | 500 000 tokenů | Multimodální | +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -> 100 % zdarma ve veřejné beta verzi. Zaregistrujte se na [longcat.chat](https://longcat.chat) pomocí e-mailu nebo telefonu. Resetuje se denně v 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -| Model | Předpona | Limit sazby | Poskytovatel za | -| ---------- | ------ | ---------- | ------------------- | -| "openai" | `pol/` | 1 požadavek/15s | GPT-5 | -| "claude" | `pol/` | 1 požadavek/15s | Antropický Claude | -| "blíženci" | `pol/` | 1 požadavek/15s | Google Gemini | -| "hluboké vyhledávání" | `pol/` | 1 požadavek/15s | DeepSeek V3 | -| "lama" | `pol/` | 1 požadavek/15s | Meta Llama 4 Scout | -| "mistrál" | `pol/` | 1 požadavek/15s | Mistral AI | +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -> ✨**Nulové tření:**Žádná registrace, žádný klíč API. Přidejte poskytovatele Pollinations s prázdným polem klíče a funguje to okamžitě.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -| Úroveň | Denní neurony | Ekvivalentní použití | Poznámky | -| ---- | ------------- | ---------------------------------------- | ------------------------ | -| Zdarma |**10 000**| ~150 LLM resp / 500s audio / 15K vložení | Global edge, 50+ modelů | +### 🔴 GROQ (Free API Key — console.groq.com) -Oblíbené bezplatné modely: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (zvuk zdarma!), `@cf/qwen/qwen2.5-coder-`1 +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ------------- | ---------------- | ----------------------------------------- | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -> Vyžaduje API Token + ID účtu z [dash.cloudflare.com](https://dash.cloudflare.com). Uložte ID účtu v nastavení poskytovatele.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -| Úroveň | Kvóta zdarma | Umístění | Poznámky | -| ---- | ------------- | ------------ | ------------------------------------ | -| Zdarma |**1 milion tokenů**| 🇫🇷 Paříž, EU | V rámci limitů není potřeba žádná kreditní karta | +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 -Dostupné zdarma: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | -> V souladu s EU/GDPR. Získejte API klíč na [console.scaleway.com](https://console.scaleway.com). +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. ->**💡 The Ultimate Free Stack (11 poskytovatelů, 0 $ navždy):** +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | +| ---------- | ------ | ---------- | ------------------ | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | + +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. + +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 + +| Tier | Daily Neurons | Equivalent Usage | Notes | +| ---- | ------------- | --------------------------------------- | ----------------------- | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | + +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` + +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. + +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | +| ---- | ------------- | ------------ | ----------------------------------- | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | + +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` + +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). + +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Sonnet/Haiku NEOMEZENO -> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -> LongCat Lite (lc/) → LongCat-Flash-Lite – 50 milionů tokenů/den 🔥 -> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — není potřeba žádný klíč -> Qwen (qw/) → modely qwen3-kodér NEOMEZENÉ -> Gemini (gemini/) → Gemini 2.5 Flash — 1 500 req/den zdarma -> Cloudflare AI (cf/) → 50+ modelů — 10 000 neuronů/den -> Scaleway (scw/) → Qwen3 235B, Llama 70B – 1M bezplatných tokenů (EU) -> Groq (groq/) → Llama/Gemma – ultrarychlé 14,4 000 požadavků/den -> NVIDIA NIM (nvidia/) → 70+ otevřených modelů — 40 RPM navždy -> Cerebras (cerebras/) → Nejrychlejší lama/Qwen na světě – 1 milion toku/den -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Přepis jakéhokoli zvuku/videa za**$0**— Deepgram vede s 200 $ zdarma, AssemblyAI 50 $ nouzové zálohy, Groq Whisper jako neomezené nouzové zálohování. +## 🎙️ Free Transcription Combo -| Poskytovatel | Kredity zdarma | Nejlepší modelka | Limit sazby | -| ------------------ | ----------------------- | --------------------------------------------- | ----------------------------- | -| 🢢**Deepgram**|**200 $ zdarma**(registrace) | `nova-3` — nejlepší přesnost, více než 30 jazyků | Žádný limit RPM na bezplatné kredity | -| 🔵**SestaveníAI**|**50 $ zdarma**(registrace) | `universal-3-pro` — kapitoly, sentiment, PII | Žádný limit RPM na bezplatné kredity | -| 🔴**Groq**|**Navždy zdarma**| `whisper-large-v3` — OpenAI Whisper | 30 RPM (rychlost omezená) | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. -**Doporučená kombinace v `/dashboard/combos`:**``` +| Provider | Free Credits | Best Model | Rate Limit | +| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | + +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Poté v `/dashboard/media` → karta**Přepis**: nahrajte jakýkoli zvukový nebo video soubor → vyberte svůj kombinovaný koncový bod → získejte přepis v podporovaných formátech.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 je postaven jako operační platforma, nikoli pouze jako přenosová proxy.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Funkce | Co to dělá | -| ---------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Grok-4 Fast Family** | xAI modely za 0,20 $/0,50 $/M – srovnávací 1143 ms (o 30 % rychlejší než Gemini 2.5 Flash) | -| 🧠**GLM-5 přes Z.AI** | 128 000 výstupní kontext, 0,5 $/1 milion – nejnovější vlajková loď z rodiny GLM | -| 🔮**MiniMax M2.5** | Úvahy + agentní úkoly za 0,30 $/1 milion – významný upgrade z M2,1 | -| 🎯**toolCalling Flag na model** | „ToolCalling: true/false“ pro model v registru – AutoCombo přeskočí modely, které nepodporují nástroje | -| 🌍**Multilingual Intent Detection** | Klíčová slova PT/ZH/ES/AR v hodnocení AutoCombo – lepší výběr modelu pro neanglický obsah | -| 📊**Zástupy založené na benchmarku** | Skutečná latence p95 z kombinovaného bodování zdrojů živých požadavků – AutoCombo se učí ze skutečných dat | -| 🔁**Požádat o deduplikaci** | Okno pro odstranění duplicitního obsahu založené na hašování obsahu – bezpečné pro více agentů, zabraňuje duplicitním poplatkům | -| 🔌**Strategie připojitelného směrovače** | Rozšiřitelné rozhraní `RouterStrategy` — přidejte vlastní logiku směrování jako zásuvné moduly | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Funkce | Co to dělá | -| -------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------- | -| 🎮**Modelové hřiště** | Stránka řídicího panelu pro přímé testování jakéhokoli modelu — voliče poskytovatele/modelu/koncového bodu, editor Monaco, streamování, přerušení, načasování | -| 🔏**CLI Fingerprint Matching** | Uspořádání záhlaví/těla podle poskytovatele tak, aby odpovídalo nativním signaturám CLI – přepněte podle poskytovatele v Nastavení > Zabezpečení.**Vaše IP adresa proxy je zachována** | -| 🤝**Podpora ACP (Protokol klienta agenta)** | Objevování agentů CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 dalších), proces spawner, koncový bod `/api/acp/agents` | -| 🤖**Hlavní panel agentů AKT** | Debug › Stránka Agenti — mřížka 14 agentů se stavem instalace, verzí, uživatelským formulářem agenta pro libovolný nástroj CLI. Uživatelé**OpenCode**získají tlačítko „Stáhnout opencode.json“, které automaticky vygeneruje konfiguraci připravenou k použití se všemi dostupnými modely. | -| 🔧**Směrování vlastního modelu `apiFormat`** | Vlastní modely s `apiFormat: "responses"` nyní správně směrují do překladače Responses API | -| 🏢**Codex Workspace Isolation** | Více pracovních prostorů Codex na e-mail — OAuth správně odděluje připojení podle ID pracovního prostoru | -| 🔄**Elektronová automatická aktualizace** | Desktopová aplikace kontroluje aktualizace + automatická instalace při restartu | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Funkce | Co to dělá | -| -------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**MCP Server (25 nástrojů)** | Nástroje IDE/agenta prostřednictvím 3 přenosů: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 jader + 3 paměti + 4 nástroje pro dovednosti | -| 🤝**Server A2A (JSON-RPC + SSE)** | Provádění úlohy agent-agent se synchronizací a streamováním | -| 🧭**Stránka konsolidovaných koncových bodů** | Stránka správy s kartami s kartami Endpoint Proxy, MCP, A2A a API Endpoints | -| 🎚️**Přepínače aktivace/deaktivace služby** | Spínače ON/OFF pro MCP a A2A s trvalým nastavením (výchozí: OFF) | -| 🛰️**MCP Runtime Heartbeat** | Skutečný stav procesu (pid, doba provozu, doba srdečního tepu, transport, režim rozsahu) | -| 📋**MCP Audit Trail** | Filtrovatelné protokoly auditu s úspěchem/neúspěchem a přiřazením klíče | -| 🔐**Vymáhání rozsahu MCP** | 10 podrobných oprávnění k rozsahu pro řízený přístup k nástrojům | -| 📡**A2A Task Lifecycle Management** | Vypsat/filtrovat úlohy, zkontrolovat události/artefakty, zrušit běžící úlohy | -| 📋**Zjištění karty agenta** | `/.well-known/agent.json` pro automatické zjišťování klienta | -| 🧪**Protokol E2E Test Harness** | Skutečný MCP SDK + klient A2A toky v `test:protocols:e2e` | -| ⚙️**Provozní ovládací prvky** | Kombinace přepínačů, použití profilů odolnosti, resetování jističů z jedné ovládací plochy | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Funkce | Co to dělá | -| ----------------------------------------------- | ----------------------------------------------------------------------------- | ----------------------- | -| 🎯**Chytrý 4úrovňový záložní zdroj** | Automatická trasa: Předplatné → Klíč API → Levné → Zdarma | -| 📊**Sledování kvót v reálném čase** | Živý počet tokenů + reset odpočítávání na poskytovatele | -| 🔄**Formátový překlad** | OpenAI ↔ Claude ↔ Gemini ↔ Odpovědi s převody bezpečnými pro schéma | -| 👥**Podpora více účtů** | Více účtů na poskytovatele s inteligentním výběrem | -| 🔄**Automatické obnovení tokenu** | Tokeny OAuth se automaticky obnovují s opakováním | -| 🎨**Vlastní kombinace** | 9 vyvažovacích strategií + řízení záložního řetězce | -| 🌐**Wildcard Router** | `poskytovatel/*` dynamické směrování | -| 🧠**Přemýšlení o kontrolách rozpočtu** | Limity průchozího, automatického, vlastního a adaptivního uvažování | -| 🔀**Aliasy modelů** | Vestavěný + vlastní model aliasing a bezpečnost migrace | -| ⚡**Degradace pozadí** | Směrujte úlohy s nízkou prioritou na pozadí na levnější modely | -| 🧪**Inteligentní směrování s ohledem na úkoly** | Automatický výběr modelu podle typu obsahu (kódování/vize/analýza/souhrn) | -| 🔄**Pracovní postupy agentů A2A** | Deterministický orchestrátor FSM pro stavové spouštění agentů ve více krocích | -| 🔀**Adaptivní směrování** | Dynamické přepisování strategie založené na objemu tokenů a složitosti výzvy | -| 🎲**Rozmanitost poskytovatelů** | Shannon entropie bodování vyvažování auto-kombo rozložení provozu | -| 💬**System Prompt Injection** | Globální ovládací prvky chování používané konzistentně | -| 📄**Kompatibilita rozhraní Responses API** | Plná podpora `/v1/responses` pro Codex a pokročilé agentní pracovní postupy | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Funkce | Co to dělá | -| --------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**Generování obrázků** | `/v1/images/generations` s cloudem a místními backendy | -| 📐**Vložení** | `/v1/embeddings` pro vyhledávání a potrubí RAG | -| 🎤**Přepis zvuku** | `/v1/audio/transscriptions` — 7 poskytovatelů (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), automatická detekce jazyka, podpora MP4/MP3/WAV | -| 🔊**Převod textu na řeč** | `/v1/audio/speech` — 10 poskytovatelů (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) se správnými chybovými zprávami | -| 🎬**Generace videa** | `/v1/videos/generations` (pracovní postupy ComfyUI + SD WebUI) | -| 🎵**Music Generation** | `/v1/music/generations` (pracovní postupy ComfyUI) | -| 🛡️**Moderování** | `/v1/moderations` bezpečnostní kontroly | -| 🔀**Reranking** | `/v1/rerank` pro hodnocení relevance | -| 🔍**Vyhledávání na webu**🆕 | `/v1/search` — 5 poskytovatelů (Serper, Brave, Perplexity, Exa, Tavily), 6 500+ zdarma/měsíc, automatické přepnutí při selhání, mezipaměť | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Funkce | Co to dělá | -| --------------------------------------- | ----------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**Jističe** | Vypnutí/obnovení pro každý model s ovládáním prahu | -| 🎯**Koncové modely** | Vlastní modely deklarují podporované koncové body + formát API | -| 🛡️**Stádo proti hromům** | Mutex + semaforové ochrany při opakování/rychlosti událostí | -| 🧠**Sémantická + mezipaměť podpisů** | Snížení nákladů/latence se dvěma vrstvami mezipaměti | -| ⚡**Žádost o idempotenci** | Duplicitní ochranné okno | -| 🔒**TLS Fingerprint Spoofing** | Otisk TLS jako v prohlížeči —**snižuje detekci robotů a nahlašování účtu** | -| 🔏**CLI Fingerprint Matching** | Odpovídá nativním podpisům požadavku CLI —**snižuje riziko zákazu při zachování proxy IP** | -| 🌐**Filtrování IP** | Kontrola seznamu povolených/blokovaných pro vystavená nasazení | -| 📊**Upravitelné limity sazeb** | Konfigurovatelné globální limity/limity na úrovni poskytovatele s perzistencí | -| 📉**Půvabná degradace** | Záložní funkce vícevrstvé ochrany chránící operace hlavní brány | -| 📜**Config Audit Trail** | Sledování změn založené na rozdílech zabraňující provoznímu posunu s jednoduchým vrácením zpět | -| ⏳**Provider Health Sync** | Proaktivní monitorování vypršení platnosti tokenu spouštějící výstrahy před selháním autorizace | -| 🚪**Automaticky zakázat zakázané účty** | Provozní jistič automaticky zaplombuje trvale zablokované tokenové účty | -| 🔑**Správa klíčů API + rozsah** | Bezpečné vydávání/otočení klíčů a ovládání modelu/poskytovatele | -| 👁️**Scoped API Key Reveal**🆕 | Přihlaste se k obnově klíčů API prostřednictvím `ALLOW_API_KEY_REVEAL` | -| 🛡️**Chráněno `/modely`** | Volitelné ověřování a skrytí poskytovatele pro katalog modelů | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Funkce | Co to dělá | -| -------------------------------------- | ---------------------------------------------------------------------- | ---------------------------- | -| 📝**Požadavek + protokolování proxy** | Úplný požadavek/odpověď a protokolování proxy | -| 📉**Streamované podrobné protokoly**🆕 | Čistě rekonstruuje datové proudy SSE do uživatelského rozhraní | -| 📋**Sjednocený panel protokolů** | Požadavek, proxy, audit a zobrazení konzoly na jedné stránce | -| 🔍**Požádejte o telemetrii** | p50/p95/p99 latence a sledování požadavků | -| 🏥**Health Dashboard** | Doba provozuschopnosti, stavy jističe, uzamčení, statistiky mezipaměti | -| 💰**Sledování nákladů** | Kontroly rozpočtu a viditelnost cen podle modelu | -| 📈**Analytické vizualizace** | Statistiky využití modelu/poskytovatele a zobrazení trendů | -| 🧪**Rámec hodnocení** | Testování zlaté sady s konfigurovatelnými strategiemi shody | -| 📡**Live Diagnostics**🆕 | Sémantické vynechání mezipaměti pro přesné kombinované živé testování | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Funkce | Co to dělá | -| ----------------------------------------- | ---------------------------------------------------------------------------- | --------------------- | -| 🌐**Nasadit kdekoli** | Localhost, VPS, Docker, cloudová prostředí | -| 🚇**Tunel Cloudflare**🆕 | Integrace rychlého tunelu jedním kliknutím z řídicího panelu | -| 🔑**Filtrování modelu klíče API** | Nativní odpověď /v1/models filtrovaná přes přiřazené kontextové role nosiče | -| ⚡**Smart Cache Bypass** | Konfigurovatelná heuristika TTL a ovládací prvky nuceného opětovného načtení | -| 🔄**Zálohování/Obnova** | Export/import a toky obnovy po havárii | -| 🧙**Průvodce onboardingem** | První spuštění průvodce nastavením | -| 🔧**CLI Tools Dashboard** | Nastavení jedním kliknutím pro oblíbené kódovací nástroje | -| 🎮**Modelové hřiště** | Otestujte libovolného poskytovatele/model/koncový bod z řídicího panelu | -| 🔏**CLI Fingerprint Toggle** | Shoda otisků prstů jednotlivých poskytovatelů v Nastavení > Zabezpečení | -| 🌐**i18n (30 jazyků)** | Plná podpora řídicího panelu + docs s pokrytím RTL | -| 🧹**Vymazat všechny modely** | Vymazání seznamu modelů jedním kliknutím v detailech poskytovatele | -| 👁️**Ovládací prvky postranního panelu**🆕 | Skrýt komponenty a integrace z Nastavení vzhledu | -| 📋**Šablony vydání** | Standardizované šablony GitHub pro chyby a funkce | -| 📂**Custom Data Directory** | Přepsání `DATA_DIR` pro umístění úložiště | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1293,103 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -Když kvóta, rychlost nebo stav selžou, OmniRoute automaticky přejde na dalšího kandidáta bez ručního přepínání.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A jsou zjistitelné v uživatelském rozhraní a dokumentech (nejsou skryté) -- Rozhraní API stavu protokolu zpřístupňují živá provozní data (`/api/mcp/*`, `/api/a2a/*`) -- Panely obsahují akce pro operace 2. dne (přepínání kombinací, resetování jističe, zrušení úkolu)#### Translator + validation workflow +#### Protocol management that is visible and operable -Oblast překladatele zahrnuje: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Hřiště**: Vyžádejte si kontroly transformace -**Chat Tester**: kompletní zpáteční cesta na žádost/odpověď -**Testovací stolice**: více případů v jednom běhu -**Live Monitor**: zobrazení dopravy v reálném čase +#### Translator + validation workflow -Plus ověření protokolu se skutečnými klienty pomocí `npm run test:protocols:e2e`. +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Reference nástrojů, konfigurace IDE a příklady klientů +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A Server README](src/lib/a2a/README.md)**— Dovednosti, metody JSON-RPC, streamování a životní cyklus úloh## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute obsahuje vestavěný hodnotící rámec pro testování kvality odezvy LLM oproti zlaté sadě. Přistupte k němu přes**Analytics → Evals**na hlavním panelu.### Built-in Golden Set +## 🧪 Evaluations (Evals) -Předinstalovaná sada „OmniRoute Golden Set“ obsahuje testovací případy pro: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Pozdravy, matematika, zeměpis, generování kódu -- Kompatibilita formátu JSON, překlad, generování markdown -- Bezpečnostní odmítnutí (škodlivý obsah), počítání, booleovská logika### Evaluation Strategies +### Built-in Golden Set -| Strategie | Popis | Příklad | -| ----------------- | ---------------------------------------------------------------------- | ------------------------------- | --- | -| "přesný" | Výstup se musí přesně shodovat | "4"" | -| "obsahuje" | Výstup musí obsahovat podřetězec (nerozlišují se malá a velká písmena) | "Paříž" | -| "regulární výraz" | Výstup musí odpovídat vzoru regulárního výrazu | `"1.*2.*3"` | -| "vlastní" | Vlastní funkce JS vrací true/false | `(výstup) => výstup.délka > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) - -🧩 Nastavení MCP (Model Context Protocol) +
+🧩 MCP Setup (Model Context Protocol) -Spusťte přenos MCP v režimu stdio:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Doporučený postup ověření: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Připojte svého MCP klienta přes stdio. -2. Spusťte `omniroute_get_health`. -3. Spusťte `omniroute_list_combos`. -4. Otevřete `/dashboard/mcp` pro potvrzení prezenčního signálu, aktivity a auditu. - -Užitečná rozhraní API pro automatizaci: +Useful APIs for automation: - `GET /api/mcp/status` - `GET /api/mcp/tools` - `GET /api/mcp/audit` -- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats` - -🤝 Nastavení A2A (Agent2Agent) + -Objevte agenta:```bash +
+🤝 A2A Setup (Agent2Agent) + +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Odeslat úkol:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` - -Správa životního cyklu: +Manage lifecycle: - `GET /api/a2a/status` - `GET /api/a2a/tasks` - `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -Provozní uživatelské rozhraní: +Operational UI: -- `/dashboard/a2a` pro pozorování úkolu/stavu/toku a akce kouře
+- `/dashboard/a2a` for task/state/stream observability and smoke actions - -🧪 End-to-end validace protokolu + -Ověřte oba protokoly se skutečnými klienty:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Tím se ověřuje: +This verifies: -- Připojení/seznam/volání klienta MCP SDK -- A2A objev/odeslat/streamovat/získat/zrušit -- Křížová kontrola dat v MCP auditu a API pro správu úloh A2A
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs - -💳 Poskytovatelé předplatného### Claude Code (Pro/Max) + + +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1402,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Tip pro profesionály:**Používejte Opus pro složité úkoly, Sonnet pro rychlost. OmniRoute sleduje kvótu na model!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1416,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Každý účet Codexu má nyní přepínače zásad v `Dashboard -> Providers`: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (ZAP/VYP): vynutit zásadu prahu 5hodinového okna. -- `Týdně` (ON/OFF): vynutit zásadu týdenního prahu okna. -- Prahové chování: když povolené okno dosáhne využití >=90 %, daný účet je přeskočen. -- Rotační chování: OmniRoute automaticky směruje na další způsobilý účet Codex. -- Resetovat chování: po uplynutí času `resetAt` poskytovatele se účet automaticky znovu stane způsobilým. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Scénáře: +Scenarios: -- `5h ON` + `Weekly ON`: účet je přeskočen, když kterékoli okno dosáhne prahové hodnoty. -- `5h VYP` + `Týdně ZAP`: účet může zablokovat pouze používání týdně. -- `5h ON` + `Týdenní OFF`: účet může zablokovat pouze 5 hodin používání. -- `resetAt` prošlo: účet automaticky znovu vstoupí do rotace (bez ručního opětovného povolení).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1441,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Nejlepší hodnota:**Obrovská bezplatná úroveň! Použijte to před placenými úrovněmi.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1456,71 +1662,91 @@ Models:
- -🔑 Poskytovatelé klíčů API### NVIDIA NIM (FREE developer access — 70+ models) +
+🔑 API Key Providers -1. Zaregistrujte se: [build.nvidia.com](https://build.nvidia.com) -2. Získejte bezplatný klíč API (včetně 1000 kreditů pro odvození) -3. Ovládací panel → Přidat poskytovatele → NVIDIA NIM: - - Klíč API: `nvapi-your-key` +### NVIDIA NIM (FREE developer access — 70+ models) -**Modely:**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` a 50+ dalších +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Tip pro profesionály:**API kompatibilní s OpenAI – bezproblémově funguje s překladem formátu OmniRoute!### DeepSeek +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -1. Zaregistrujte se: [platform.deepseek.com](https://platform.deepseek.com) -2. Získejte API klíč -3. Ovládací panel → Přidat poskytovatele → DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -**Modely:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +### DeepSeek -1. Zaregistrujte se: [console.groq.com](https://console.groq.com) -2. Získejte klíč API (včetně bezplatné úrovně) -3. Ovládací panel → Přidat poskytovatele → Groq +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -**Modely:**`groq/lama-3.3-70b`, `groq/mixtral-8x7b` +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Tip pro profesionály:**Ultra rychlé vyvozování – nejlepší pro kódování v reálném čase!### OpenRouter (100+ Models) +### Groq (Free Tier Available!) -1. Zaregistrujte se: [openrouter.ai](https://openrouter.ai) -2. Získejte API klíč -3. Ovládací panel → Přidat poskytovatele → OpenRouter +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -**Modely:**Získejte přístup k více než 100 modelům od všech hlavních poskytovatelů prostřednictvím jediného klíče API. +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Chování řídicího panelu:**Modely OpenRouter jsou spravovány z**Dostupných modelů**. Ruční přidání, import a automatická synchronizace aktualizují stejný seznam.
+**Pro Tip:** Ultra-fast inference — best for real-time coding! - -💰 Levní poskytovatelé (záložní)### GLM-4.7 (Daily reset, $0.6/1M) +### OpenRouter (100+ Models) -1. Zaregistrujte se: [Zhipu AI](https://open.bigmodel.cn/) -2. Získejte API klíč z Coding Plan -3. Ovládací panel → Přidat klíč API: - - Poskytovatel: `glm` - - Klíč API: `váš klíč` +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -**Použití:**`glm/glm-4.7` +**Models:** Access 100+ models from all major providers through a single API key. -**Tip pro profesionály:**Kódovací plán nabízí 3× kvótu za 1/7 cenu! Resetovat denně v 10:00.### MiniMax M2.1 (5h reset, $0.20/1M) +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -1. Zaregistrujte se: [MiniMax](https://www.minimax.io/) -2. Získejte API klíč -3. Ovládací panel → Přidat klíč API + -**Použití:**`minimax/MiniMax-M2.1` +
+💰 Cheap Providers (Backup) -**Tip pro profesionály:**Nejlevnější možnost pro dlouhý kontext (1 milion tokenů)!### Kimi K2 ($9/month flat) +### GLM-4.7 (Daily reset, $0.6/1M) -1. Přihlaste se k odběru: [Moonshot AI](https://platform.moonshot.ai/) -2. Získejte API klíč -3. Ovládací panel → Přidat klíč API +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**Použijte:**`kimi/kimi-latest` +**Use:** `glm/glm-4.7` -**Tip pro profesionály:**Pevná cena 9 $ měsíčně za 10 milionů tokenů = 0,90 $ / 1 milion efektivních nákladů!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. - -🆓 ZDARMA poskytovatelé (nouzové zálohování)### Qoder (5 FREE models via OAuth) +### MiniMax M2.1 (5h reset, $0.20/1M) + +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `minimax/MiniMax-M2.1` + +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1561,8 +1787,10 @@ Models:
- -🎨 Vytvořit komba### Example 1: Maximize Subscription → Cheap Backup +
+🎨 Create Combos + +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1590,8 +1818,10 @@ Cost: $0 forever!
- -🔧 Integrace CLI### Cursor IDE +
+🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1602,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Použijte stránku**CLI Tools**na řídicím panelu pro konfiguraci jedním kliknutím nebo upravte `~/.claude/settings.json` ručně.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1613,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Možnost 1 – Hlavní panel (doporučeno):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Možnost 2 — Ručně:**Upravit `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1630,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Poznámka:**OpenClaw funguje pouze s místní OmniRoute. Použijte `127.0.0.1` místo `localhost`, abyste se vyhnuli problémům s rozlišením IPv6.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1644,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Krok 1:**Přidejte OmniRoute jako vlastního poskytovatele:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Krok 2:**Vytvořte/upravte soubor `opencode.json` v kořenovém adresáři projektu:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1670,118 +1909,130 @@ opencode } } } -```` +``` -**Krok 3:**Vyberte model v OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Tip:**Přidejte jakýkoli model dostupný v koncovém bodu vašeho OmniRoute `/v1/models` do sekce `models`. Použijte formát `provider/model-id` z řídicího panelu OmniRoute.
+ --- ## Řešení problémů - -Kliknutím rozbalíte průvodce odstraňováním problémů +
+Click to expand troubleshooting guide -**"Jazykový model neposkytoval zprávy"** +**"Language model did not provide messages"** -- Kvóta poskytovatele je vyčerpána → Zkontrolujte sledování kvót na řídicím panelu -- Řešení: Použijte nouzovou kombinaci nebo přejděte na levnější úroveň +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**Omezení sazby** +**Rate limiting** -- Vyčerpaná kvóta předplatného → Záložní režim GLM/MiniMax -- Přidejte kombinaci: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**Platnost tokenu OAuth vypršela** +**OAuth token expired** -- Automaticky obnovováno OmniRoute -- Pokud problémy přetrvávají: Řídicí panel → Poskytovatel → Znovu připojit +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Vysoké náklady** +**High costs** -- Zkontrolujte statistiky využití v Dashboard → Náklady -- Přepněte primární model na GLM/MiniMax -- Používejte bezplatnou vrstvu (Gemini CLI, Qoder) pro nekritické úkoly +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Porty řídicího panelu/API jsou chybné** +**Dashboard/API ports are wrong** -- `PORT` je kanonický základní port (a port API ve výchozím nastavení) -- `API_PORT` přepíše pouze posluchače API kompatibilní s OpenAI -- `DASHBOARD_PORT` přepíše pouze posluchače dashboard/Next.js -– Nastavte „NEXT_PUBLIC_BASE_URL“ na svůj řídicí panel/veřejnou adresu URL (pro zpětná volání OAuth) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Chyby synchronizace cloudu** +**Cloud sync errors** -- Ověřte, že `BASE_URL` odkazuje na vaši spuštěnou instanci -- Ověřte, že `CLOUD_URL` odkazuje na očekávaný koncový bod cloudu -- Udržujte hodnoty `NEXT_PUBLIC_*` zarovnané s hodnotami na straně serveru +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**První přihlášení nefunguje** +**First login not working** -- Zkontrolujte `INITIAL_PASSWORD` v `.env` -- Pokud není nastaveno, záložní heslo je `123456` +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Žádné protokoly požadavků** +**No request logs** -- Artefakty požadavku se zapisují do `DATA_DIR/call_logs/` jako jeden soubor JSON na požadavek -- Povolte zachycení potrubí z řídicího panelu → Protokoly → Protokoly žádostí, pokud potřebujete podrobné užitečné zatížení pro jednotlivé fáze -- Nastavte `APP_LOG_TO_FILE=true`, pokud chcete také protokoly konzoly aplikace v `logs/application/app.log` -– Podle potřeby upravte `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` a `CALL_LOG_MAX_ENTRIES` +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Test připojení ukazuje „Neplatné“ pro poskytovatele kompatibilní s OpenAI** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Mnoho poskytovatelů nevystavuje koncový bod `/models` -- OmniRoute v1.0.6+ zahrnuje nouzové ověření prostřednictvím dokončení chatu -- Zajistěte, aby základní adresa URL obsahovala příponu `/v1`### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ Důležité pro uživatele provozující OmniRoute na VPS, Dockeru nebo jakémkoli vzdáleném serveru**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -Poskytovatelé**Antigravity**a**Gemini CLI**používají**Google OAuth 2.0**. Google vyžaduje, aby parametr `redirect_uri` v toku OAuth přesně odpovídal jednomu z předem registrovaných URI v Google Cloud Console aplikace. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -Přihlašovací údaje OAuth dodávané v OmniRoute jsou registrovány**pouze pro `localhost`**. Když přistupujete k OmniRoute na vzdáleném serveru (např. `https://omniroute.myserver.com`), Google odmítne ověření pomocí:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Ve službě Google Cloud Console musíte vytvořit**OAuth 2.0 Client ID**s identifikátorem URI vašeho serveru.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Otevřít Google Cloud Console** +#### Step-by-step -Přejděte na: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Vytvořit nové ID klienta OAuth 2.0** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -– Klikněte na**"+ Vytvořit přihlašovací údaje"**→**"ID klienta OAuth"** +**2. Create a new OAuth 2.0 Client ID** -- Typ aplikace:**"Webová aplikace"** -- Název: cokoliv se vám líbí (např. `OmniRoute Remote`) +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -**3. Přidat identifikátory URI autorizovaného přesměrování** +**3. Add Authorized Redirect URIs** -Do pole**"URI autorizovaného přesměrování"**přidejte:``` +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Nahraďte `vas-server.com` doménou nebo IP svého serveru (v případě potřeby uveďte port, např. `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. Uložte a zkopírujte přihlašovací údaje** +After creating, Google will show the **Client ID** and **Client Secret**. -Po vytvoření Google zobrazí**Client ID**a**Client Secret**. +**5. Set environment variables** -**5. Nastavit proměnné prostředí** +In your `.env` (or Docker environment variables): -Ve vašem `.env` (nebo proměnných prostředí Docker):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1790,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Restartujte OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Zkuste se připojit znovu** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Ovládací panel → Poskytovatelé → Antigravitace (nebo Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google se nyní správně přesměruje na `https://vas-server.com/callback`.--- +--- #### Temporary workaround (without custom credentials) -Pokud si nyní nechcete nastavovat vlastní přihlašovací údaje, můžete stále použít**ruční postup URL**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute otevře autorizační URL Google -2. Po autorizaci se Google pokusí přesměrovat na `localhost` (který selže na vzdáleném serveru) -3.**Zkopírujte celou adresu URL**z adresního řádku prohlížeče (i když se stránka nenačte) -4. Vložte tuto adresu URL do pole zobrazeného v modálu připojení OmniRoute -5. Klikněte na**"Připojit"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Funguje to, protože autorizační kód v adrese URL je platný bez ohledu na to, zda se stránka přesměrování načetla.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. - -🇧🇷 Versão em Português#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Osvedčuje**Antigravity**a**Gemini CLI**používáme**Google OAuth 2.0**pro autenticitu. O Google exige que a `redirect_uri` usada no fluxo OAuth seja**exatamente**uma das URIs pré-cadastradas no Google Cloud Console to use. +
+🇧🇷 Versão em Português -Jako credenciais OAuth embutidas no OmniRoute estão cadastradas**apenas para `localhost`**. Quando você acessa o OmniRoute um um servidor remote (ex: `https://omniroute.meuservidor.com`), nebo Google rejeita a autenticação com:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Você precisa criar um**OAuth 2.0 Client ID**no Google Cloud Console com a URI do seu server.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Přístup ke službě Google Cloud Console** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) **2. Crie um novo OAuth 2.0 Client ID** -- Klikněte na**"+ Vytvořit přihlašovací údaje"**→**"ID klienta OAuth"** -- Tipo de aplicativo:**"Webová aplikace"** -- Nome: escolha qualquer nome (např.: `OmniRoute Remote`) +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Adicione jako Authorized Redirect URI** +**3. Adicione as Authorized Redirect URIs** -Žádné pole**"URI autorizovaného přesměrování"**, adicione:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Substitua `seu-servidor.com` pelo domínio nebo IP do seu servidor (včetně portu se necessário, např.: `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. Uložit a zkopírovat jako credenciais** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Após criar, o Google mostrará o**Client ID**e o**Client Secret**. +**5. Configure as variáveis de ambiente** -**5. Konfigurovat jako variáveis de ambiente** +No seu `.env` (ou nas variáveis de ambiente do Docker): -No seu `.env` (ou nas variáveis de ambiente do Docker):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1869,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reinicie nebo OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute - -```` +``` **7. Tente conectar novamente** -Dashboard → Poskytovatelé → Antigravitace (nebo Gemini CLI) → OAuth +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Agora nebo Google redirecionamente corretamente para `https://seu-servidor.com/callback` a autenticação funcionará.--- +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. + +--- #### Workaround temporário (sem configurar credenciais próprias) -Zjistěte, jaké jsou údaje o vaší kreditní kartě, a je možné, že použijete fluxo**příručku URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. O OmniRoute abrirá a URL autorização Google +1. O OmniRoute abrirá a URL de autorização do Google 2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) -3.**Zkopírujte úplnou adresu URL**da barra de endereço do seu browser (mesmo que a pagina não carregue) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) 4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute -5. Klikněte na**"Připojit"** +5. Clique em **"Connect"** -> Toto řešení funguje pomocí autorizačního kódu na URL a nezávislého přesměrování.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1907,64 +2171,73 @@ Zjistěte, jaké jsou údaje o vaší kreditní kartě, a je možné, že použi ## 🛠️ Tech Stack - -Kliknutím rozbalíte podrobnosti o technologickém zásobníku +
+Click to expand tech stack details --**Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ není**podporován**– nativní binární soubory `better-sqlite3` jsou nekompatibilní) --**Jazyk**: TypeScript 5.9 —**100% TypeScript**napříč `src/` a `open-sse/` (nula `any` v základních modulech od verze 2.0) --**Framework**: Next.js 16 + React 19 + Tailwind CSS 4 --**Databáze**: LowDB (JSON) + SQLite (stav domény + protokoly proxy + audit MCP + rozhodnutí o směrování) --**Schémata**: Zod (ověření I/O nástroje MCP, smlouvy API) --**Protokoly**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Streamování**: Server-Sent Events (SSE) --**Auth**: OAuth 2.0 (PKCE) + JWT + API klíče + MCP Scoped Authorization --**Testování**: Testovací program Node.js + Vitest (více než 900 testů včetně jednotky, integrace, E2E) --**CI/CD**: Akce GitHub (automatické publikování npm + Docker Hub při vydání) --**Web**: [omniroute.online](https://omniroute.online) --**Balík**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Odolnost**: Jistič, exponenciální ústup, stádo proti hromům, TLS spoofing, auto-kombo samoléčení
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Dokumentace -| Dokument | Popis | -| ---------------------------------------------- | ---------------------------------------------------- | -| [Uživatelská příručka](docs/USER_GUIDE.md) | Poskytovatelé, komba, integrace CLI, nasazení | -| [Reference API](docs/API_REFERENCE.md) | Všechny koncové body s příklady | -| [Server MCP](open-sse/mcp-server/README.md) | 16 MCP nástroje, konfigurace IDE, klienti Python/TS/Go | -| [Server A2A](src/lib/a2a/README.md) | Protokol JSON-RPC 2.0, dovednosti, streamování, správa úloh | -| [Auto-Combo Engine](docs/auto-combo.md) | 6faktorové bodování, balíčky režimů, samoléčení | -| [Odstraňování problémů](docs/TROUBLESHOOTING.md) | Běžné problémy a řešení | -| [Architektura](docs/ARCHITECTURE.md) | Architektura systému a vnitřní části | -| [Přispívá](CONTRIBUTING.md) | Vývojové nastavení a pokyny | -| [Specifikace OpenAPI](docs/openapi.yaml) | Specifikace OpenAPI 3.0 | -| [Bezpečnostní zásady](SECURITY.md) | Hlášení zranitelnosti a bezpečnostní postupy | -| [Deployment VM](docs/VM_DEPLOYMENT_GUIDE.md) | Kompletní průvodce: Nastavení VM + nginx + Cloudflare | -| [Galerie funkcí](docs/FEATURES.md) | Vizuální prohlídka řídicího panelu se snímky obrazovky | -| [Kontrolní seznam vydání](docs/RELEASE_CHECKLIST.md) | Kroky ověření před vydáním |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute má naplánováno**210+ funkcí**v několika fázích vývoje. Zde jsou klíčové oblasti: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Kategorie | Plánované funkce | Hlavní body | -| ------------------------------ | ----------------- | -------------------------------------------------------------------------------------- | -| 🧠**Směrování a inteligence**| 25+ | Směrování s nejnižší latencí, směrování založené na značkách, předběžná kontrola kvót, výběr účtu P2C | -| 🔒**Zabezpečení a dodržování předpisů**| 20+ | Zpevnění SSRF, maskování pověření, rychlostní limit na koncový bod, stanovení rozsahu klíče managementu | -| 📊**Pozorovatelnost**| 15+ | Integrace OpenTelemetry, sledování kvót v reálném čase, sledování nákladů na model | -| 🔄**Integrace poskytovatelů**| 20+ | Registr dynamického modelu, cooldowny poskytovatelů, kodex pro více účtů, analýza kvót Copilota | -| ⚡**Výkon**| 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | -| 🌐**Ekosystém**| 10+ | WebSocket API, konfigurace hot-reload, distribuované úložiště konfigurace, komerční režim |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**Integrace OpenCode**– Podpora nativního poskytovatele pro IDE kódování OpenCode AI -- 🔗**Integrace TRAE**— Plná podpora pro vývojový rámec TRAE AI -- 📦**Batch API**— Asynchronní dávkové zpracování pro hromadné požadavky -- 🎯**Směrování založené na značkách**– Směrování požadavků na základě vlastních značek a metadat -- 💰**Strategie nejnižších nákladů**— Automaticky vyberte nejlevnějšího dostupného poskytovatele +### 🔜 Coming Soon -> 📝 Úplné specifikace funkcí dostupné v [`docs/new-features/`](docs/new-features/) (217 podrobných specifikací)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1972,18 +2245,20 @@ OmniRoute má naplánováno**210+ funkcí**v několika fázích vývoje. Zde jso ### How to Contribute -1. Rozdělte úložiště -2. Vytvořte si větev funkcí (`git checkout -b feature/amazing-feature`) -3. Potvrďte změny (`git commit -m 'Přidat úžasnou funkci'`) -4. Push do větve (`git Push origin feature/amazing-feature`) -5. Otevřete žádost o stažení +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Podrobné pokyny najdete na [CONTRIBUTING.md](CONTRIBUTING.md).### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1995,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Zvláštní poděkování patří**[9router](https://github.com/decolua/9router)**od**[decolua](https://github.com/decolua)**— původnímu projektu, který inspiroval tento fork. OmniRoute staví na tomto neuvěřitelném základu s dalšími funkcemi, multimodálními API a úplným přepsáním TypeScriptu. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Zvláštní poděkování patří**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**– původní implementaci Go, která inspirovala tento port JavaScriptu.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Licence -Licence MIT – podrobnosti viz [LICENCE](LICENCE).--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/cs/docs/ARCHITECTURE.md b/docs/i18n/cs/docs/ARCHITECTURE.md index e6d8fdc60c..587e9fe20c 100644 --- a/docs/i18n/cs/docs/ARCHITECTURE.md +++ b/docs/i18n/cs/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Poslední aktualizace: 28.03.2026_## Executive Summary -OmniRoute je místní AI směrovací brána a řídicí panel postavený na Next.js. -Poskytuje jeden koncový bod kompatibilní s OpenAI (`/v1/*`) a směruje provoz přes několik upstreamových poskytovatelů s překladem, nouzovým obnovením, obnovením tokenu a sledováním využití. -Základní schopnosti: +_Last updated: 2026-03-28_ -- OpenAI kompatibilní povrch API pro CLI/nástroje (28 poskytovatelů) -- Překlad požadavku/odpovědi mezi formáty poskytovatelů -- Záložní kombinace modelů (sekvence více modelů) - – Záloha na úrovni účtu (více účtů na poskytovatele) -- Správa připojení poskytovatele OAuth + API klíče -- Generování vkládání pomocí `/v1/embeddings` (6 poskytovatelů, 9 modelů) -- Generování obrázků pomocí `/v1/images/generations` (4 poskytovatelé, 9 modelů) -- Myslete na analýzu značek (`...`) pro modely uvažování -- Dezinfekce odezvy pro přísnou kompatibilitu OpenAI SDK -- Normalizace rolí (vývojář→systém, systém→uživatel) pro kompatibilitu mezi poskytovateli -- Konverze strukturovaného výstupu (json_schema → Gemini responseSchema) -- Místní perzistence pro poskytovatele, klíče, aliasy, komba, nastavení, ceny -- Sledování využití/nákladů a protokolování požadavků -- Volitelná cloudová synchronizace pro synchronizaci mezi více zařízeními/stavy -- Seznam povolených/blokovaných IP pro řízení přístupu k API -- Myslet na správu rozpočtu (průchozí/automatické/vlastní/adaptivní) -- Okamžité vstřikování globálního systému -- Sledování relací a snímání otisků prstů -- Rozšířené omezení sazeb na účet pomocí profilů specifických pro poskytovatele -- Vzor jističe pro odolnost poskytovatele -- Ochrana stáda proti hromu s mutexovým zamykáním -- Mezipaměť deduplikace požadavků na základě podpisu -- Doménová vrstva: dostupnost modelu, nákladová pravidla, záložní politika, politika uzamčení -- Perzistence stavu domény (mezipaměť pro zápis SQLite pro záložní, rozpočty, uzamčení, jističe) -- Modul zásad pro centralizované vyhodnocování požadavků (uzamčení → rozpočet → záložní) -- Vyžádejte si telemetrii s agregací latence p50/p95/p99 -- ID korelace (X-Request-Id) pro end-to-end trasování -- Protokolování auditu shody s odhlášením podle klíče API -- Eval rámec pro zajištění kvality LLM -- Řídicí panel Resilience UI se stavem jističe v reálném čase -- Modulární poskytovatelé OAuth (12 jednotlivých modulů pod `src/lib/oauth/providers/`) +## Executive Summary -Primární runtime model: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- Trasy aplikací Next.js pod `src/app/api/*` implementují jak rozhraní API řídicího panelu, tak rozhraní API pro kompatibilitu -- Sdílené jádro SSE/směrování v `src/sse/*` + `open-sse/*` se stará o provádění poskytovatele, překlad, streamování, zálohování a používání## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Runtime místní brány -- Rozhraní API pro správu řídicích panelů -- Ověření poskytovatele a obnovení tokenu -- Vyžádejte si překlad a streamování SSE -- Místní stav + perzistence používání -- Volitelná orchestrace synchronizace s cloudem### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Implementace cloudové služby za `NEXT_PUBLIC_CLOUD_URL` -- Poskytovatel SLA/řídící rovina mimo místní proces -- Samotné externí binární soubory CLI (Claude CLI, Codex CLI atd.)## Dashboard Surface (Current) +### Out of Scope -Hlavní stránky pod `src/app/(dashboard)/dashboard/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — rychlý start + přehled poskytovatele -- `/dashboard/endpoint` — proxy koncového bodu + MCP + A2A + karty koncového bodu API -- `/dashboard/providers` — připojení a přihlašovací údaje poskytovatele -- `/dashboard/combos` — kombo strategie, šablony, pravidla směrování modelů -- `/dashboard/costs` — agregace nákladů a viditelnost cen -- `/dashboard/analytics` — analýzy a vyhodnocení využití -- `/dashboard/limits` — kontroly kvót/sazeb -- `/dashboard/cli-tools` — CLI onboarding, runtime detekce, generování konfigurace -- `/dashboard/agents` — detekovaní agenti AKT + vlastní registrace agenta -- `/dashboard/media` — hřiště pro obrázky/video/hudbu -- `/dashboard/search-tools` — testování a historie poskytovatelů vyhledávání -- `/dashboard/health` — doba provozuschopnosti, jističe, limity sazeb -- `/dashboard/logs` — protokoly požadavku/proxy/audit/konzole -- `/dashboard/settings` — karty nastavení systému (obecné, směrování, výchozí kombinace atd.) -- `/dashboard/api-manager` — životní cyklus klíče API a oprávnění k modelu## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Hlavní adresáře: +Main directories: -- `src/app/api/v1/*` a `src/app/api/v1beta/*` pro rozhraní API pro kompatibilitu -- `src/app/api/*` pro správu/konfiguraci API -- Další přepíše mapu `next.config.mjs` `/v1/*` na `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Důležité cesty kompatibility: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` — zahrnuje vlastní modely s `custom: true` -- `src/app/api/v1/embeddings/route.ts` — generování vložení (6 poskytovatelů) -- `src/app/api/v1/images/generations/route.ts` — generování obrázků (4+ poskytovatelé včetně Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` - – `src/app/api/v1/providers/[poskytovatel]/chat/completions/route.ts` – vyhrazený chat pro jednotlivé poskytovatele - – `src/app/api/v1/providers/[poskytovatel]/embeddings/route.ts` – vyhrazená vložení pro jednotlivé poskytovatele -- `src/app/api/v1/providers/[poskytovatel]/images/generations/route.ts` – vyhrazené obrázky pro jednotlivé poskytovatele +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` -- `src/app/api/v1beta/models/[...cesta]/route.ts` +- `src/app/api/v1beta/models/[...path]/route.ts` -Domény správy: +Management domains: - Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` -- Poskytovatelé/připojení: `src/app/api/providers*` -- Uzly poskytovatele: `src/app/api/provider-nodes*` -- Vlastní modely: `src/app/api/provider-models` (GET/POST/DELETE) -- Katalog modelů: `src/app/api/models/route.ts` (GET) -- Konfigurace proxy: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- Klíče/aliasy/komba/cena: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- Použití: `src/app/api/usage/*` +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` - Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` -- Pomocníci nástrojů CLI: `src/app/api/cli-tools/*` -- IP filtr: `src/app/api/settings/ip-filter` (GET/PUT) +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) - Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) -- Systémová výzva: `src/app/api/settings/system-prompt` (GET/PUT) -- Relace: `src/app/api/sessions` (GET) -- Sazbové limity: `src/app/api/rate-limits` (GET) -- Odolnost: `src/app/api/resilience` (GET/PATCH) — profily poskytovatelů, jistič, stav omezení rychlosti -- Resetování odolnosti: `src/app/api/resilience/reset` (POST) — resetujte jističe + cooldowny -- Statistiky mezipaměti: `src/app/api/cache/stats` (GET/DELETE) -- Dostupnost modelu: `src/app/api/models/availability` (GET/POST) -- Telemetrie: `src/app/api/telemetry/summary` (GET) - – Rozpočet: `src/app/api/usage/budget` (GET/POST) -- Záložní řetězce: `src/app/api/fallback/chains` (GET/POST/DELETE) -- Audit souladu: `src/app/api/compliance/audit-log` (GET) -- Hodnoty: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- Zásady: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -Hlavní průtokové moduly: +## 2) SSE + Translation Core -- Záznam: `src/sse/handlers/chat.ts` -- Základní orchestrace: `open-sse/handlers/chatCore.ts` -- Spouštěcí adaptéry poskytovatele: `open-sse/executors/*` -- Detekce formátu/konfigurace poskytovatele: `open-sse/services/provider.ts` -- Parse/resolve modelu: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Logika záložního účtu: `open-sse/services/accountFallback.ts` -- Registr překladů: `open-sse/translator/index.ts` -- Transformace streamu: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- Extrakce/normalizace použití: `open-sse/utils/usageTracking.ts` -- Analyzátor značek Think: `open-sse/utils/thinkTagParser.ts` -- Obsluha vkládání: `open-sse/handlers/embeddings.ts` -- Registr poskytovatele vkládání: `open-sse/config/embeddingRegistry.ts` -- Ovladač generování obrázků: `open-sse/handlers/imageGeneration.ts` -- Registr poskytovatele obrázků: `open-sse/config/imageRegistry.ts` -- Dezinfekce odezvy: `open-sse/handlers/responseSanitizer.ts` -- Normalizace rolí: `open-sse/services/roleNormalizer.ts` +Main flow modules: -Služby (obchodní logika): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- Výběr účtu/bodování: `open-sse/services/accountSelector.ts` -- Kontextová správa životního cyklu: `open-sse/services/contextManager.ts` -- Vynucení filtru IP: `open-sse/services/ipFilter.ts` -- Sledování relací: `open-sse/services/sessionManager.ts` -- Žádost o deduplikaci: `open-sse/services/signatureCache.ts` -- Vložení příkazu systému: `open-sse/services/systemPrompt.ts` +Services (business logic): + +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` - Thinking budget management: `open-sse/services/thinkingBudget.ts` -- Směrování modelu se zástupnými znaky: `open-sse/services/wildcardRouter.ts` -- Správa limitu sazby: `open-sse/services/rateLimitManager.ts` -- Jistič: `open-sse/services/circuitBreaker.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -Moduly vrstvy domény: +Domain layer modules: -- Dostupnost modelu: `src/lib/domain/modelAvailability.ts` -- Pravidla/rozpočty nákladů: `src/lib/domain/costRules.ts` -- Záložní zásady: `src/lib/domain/fallbackPolicy.ts` +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` - Combo resolver: `src/lib/domain/comboResolver.ts` -- Zásady uzamčení: `src/lib/domain/lockoutPolicy.ts` -- Modul zásad: `src/domain/policyEngine.ts` — centralizované uzamčení → rozpočet → záložní vyhodnocení -- Katalog chybových kódů: `src/lib/domain/errorCodes.ts` -- ID požadavku: `src/lib/domain/requestId.ts` -- Časový limit načtení: `src/lib/domain/fetchTimeout.ts` -- Žádost o telemetrii: `src/lib/domain/requestTelemetry.ts` -- Soulad/audit: `src/lib/domain/compliance/index.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` - Eval runner: `src/lib/domain/evalRunner.ts` -- Trvalost stavu domény: `src/lib/db/domainState.ts` — SQLite CRUD pro záložní řetězce, rozpočty, historii nákladů, stav uzamčení, jističe +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -Moduly poskytovatele OAuth (12 samostatných souborů pod `src/lib/oauth/providers/`): +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -- Index registru: `src/lib/oauth/providers/index.ts` - – Jednotliví poskytovatelé: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilo`codes`, `kilo`code -- Tenký obal: `src/lib/oauth/providers.ts` — reexporty z jednotlivých modulů## 3) Persistence Layer +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -Primární stav DB (SQLite): +## 3) Persistence Layer -- Základní jádro: `src/lib/db/core.ts` (better-sqlite3, migrace, WAL) -- Fasáda pro reexport: `src/lib/localDb.ts` (tenká vrstva kompatibility pro volající) -- soubor: `${DATA_DIR}/storage.sqlite` (nebo `$XDG_CONFIG_HOME/omniroute/storage.sqlite`, pokud je nastaven, jinak `~/.omniroute/storage.sqlite`) -- entity (tabulky + jmenné prostory KV): providerConnections, providerNodes, modelAliases, komba, apiKeys, nastavení, ceny,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +Primary state DB (SQLite): -Perzistence při používání: +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -- fasáda: `src/lib/usageDb.ts` (rozložené moduly v `src/lib/usage/*`) -- SQLite tabulky v `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` -- volitelné artefakty souborů zůstávají kvůli kompatibilitě/ladění (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- starší soubory JSON jsou migrovány do SQLite migrací při spuštění, pokud jsou k dispozici +Usage persistence: -Stavová databáze domény (SQLite): +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- `src/lib/db/domainState.ts` — operace CRUD pro stav domény - – Tabulky (vytvořené v `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- Vzor mezipaměti pro zápis: mapy v paměti jsou autoritativní za běhu; mutace se zapisují synchronně do SQLite; stav je obnoven z DB při studeném startu## 4) Auth + Security Surfaces +Domain State DB (SQLite): -- Ověření souboru cookie řídicího panelu: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- Generování/ověření klíče API: `src/shared/utils/apiKey.ts` -- Tajné informace poskytovatele zůstaly v položkách `providerConnections` -- Podpora odchozích proxy přes `open-sse/utils/proxyFetch.ts` (env vars) a `open-sse/utils/networkProxy.ts` (konfigurovatelné pro jednotlivé poskytovatele nebo globální)## 5) Cloud Sync +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start -- Init plánovače: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Pravidelný úkol: `src/shared/services/cloudSyncScheduler.ts` -- Pravidelný úkol: `src/shared/services/modelSyncScheduler.ts` -- Řídící cesta: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Záložní rozhodnutí jsou řízena `open-sse/services/accountFallback.ts` pomocí stavových kódů a heuristiky chybových zpráv. Kombinované směrování přidává ještě jednu ochranu: 400s v rozsahu poskytovatele, jako jsou selhání blokování obsahu a ověřování rolí, jsou považovány za lokální selhání modelu, takže pozdější kombinované cíle mohou stále běžet.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -Obnovení během živého provozu se provádí uvnitř `open-sse/handlers/chatCore.ts` prostřednictvím spouštěče `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -Pravidelnou synchronizaci spouští „CloudSyncScheduler“, když je povolen cloud.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -Soubory fyzického úložiště: +Physical storage files: -- primární runtime DB: `${DATA_DIR}/storage.sqlite` -- řádky protokolu požadavku: `${DATA_DIR}/log.txt` (artefakt compat/debug) -- archivy strukturovaného obsahu volání: `${DATA_DIR}/call_logs/` -- volitelné relace ladění překladatele/požadavku: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: rozhraní API pro kompatibilitu -- `src/app/api/v1/providers/[poskytovatel]/*`: vyhrazené trasy pro jednotlivé poskytovatele (chat, vkládání, obrázky) -- `src/app/api/providers*`: poskytovatel CRUD, ověření, testování -- `src/app/api/provider-nodes*`: vlastní kompatibilní správa uzlů -- `src/app/api/provider-models`: správa vlastních modelů (CRUD) -- `src/app/api/models/route.ts`: API katalogu modelů (aliasy + vlastní modely) -- `src/app/api/oauth/*`: toky OAuth/kódu zařízení -- `src/app/api/keys*`: životní cyklus místního klíče API -- `src/app/api/models/alias`: správa aliasů -- `src/app/api/combos*`: správa záložních kombinací -- `src/app/api/pricing`: přepisy cen pro výpočet nákladů -- `src/app/api/settings/proxy`: konfigurace proxy (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: test odchozího proxy připojení (POST) -- `src/app/api/usage/*`: využití a protokoly API -- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloudová synchronizace a pomocníci pro cloud -- `src/app/api/cli-tools/*`: místní zapisovače/kontroly konfigurace CLI -- `src/app/api/settings/ip-filter`: seznam povolených/blokovaných IP adres (GET/PUT) -- `src/app/api/settings/thinking-budget`: konfigurace rozpočtu tokenu myšlení (GET/PUT) -- `src/app/api/settings/system-prompt`: globální systémová výzva (GET/PUT) -- `src/app/api/sessions`: seznam aktivních relací (GET) -- `src/app/api/rate-limits`: stav limitu sazby na účet (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: analýza požadavků, zpracování kombinací, smyčka výběru účtu -- `open-sse/handlers/chatCore.ts`: překlad, odeslání exekutora, zpracování opakování/obnovení, nastavení streamu -- `open-sse/executors/*`: chování sítě a formátu specifické pro poskytovatele### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: registr a orchestrace překladatelů -- Požadavek na překladatele: `open-sse/translator/request/*` -- Překladače odpovědí: `open-sse/translator/response/*` -- Formátové konstanty: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: trvalá trvalá konfigurace/stav a doména na SQLite -- `src/lib/localDb.ts`: reexport kompatibility pro moduly DB -- `src/lib/usageDb.ts`: fasáda historie použití/protokolů volání nad tabulkami SQLite## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Každý poskytovatel má specializovaný spouštěč rozšiřující `BaseExecutor` (v `open-sse/executors/base.ts`), který poskytuje vytváření URL, konstrukci záhlaví, opakování s exponenciálním stažením, háky pro obnovení pověření a metodu orchestrace `execute()`. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Exekutor | Poskytovatel(é) | Speciální manipulace | -| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ---------------------------------------------------------------------------------------- | -| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamická konfigurace URL/záhlaví na poskytovatele | -| "AntigravityExecutor" | Google Antigravity | Vlastní ID projektů/relací, Opakovat po analýze | -| "CodexExecutor" | Kodex OpenAI | Vkládá systémové instrukce, nutí k logickému úsilí | -| `CursorExecutor` | Kurzor IDE | Protokol ConnectRPC, kódování Protobuf, podepisování požadavků pomocí kontrolního součtu | -| "GithubExecutor" | GitHub Copilot | Obnovení tokenu druhého pilota, hlavičky napodobující VSCode | -| "KiroExecutor" | AWS CodeWhisperer/Kiro | Binární formát AWS EventStream → konverze SSE | -| "GeminiCLIExecutor" | Gemini CLI | Cyklus obnovení tokenu Google OAuth | +### Persistence -Všichni ostatní poskytovatelé (včetně vlastních kompatibilních uzlů) používají `DefaultExecutor`.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Poskytovatel | Formát | Auth | Stream | Nestreamovat | Obnovení tokenu | Použití API | -| ---------------- | ---------------- | ------------------------ | -------------------- | ------------ | --------------- | ------------------- | ------------------------------ | -| Claude | claude | Klíč API / OAuth | ✅ | ✅ | ✅ | ⚠️ Pouze správce | -| Blíženci | Blíženci | Klíč API / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloudová konzole | -| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloudová konzole | -| Antigravitace | antigravitace | OAuth | ✅ | ✅ | ✅ | ✅ Plná kvóta API | -| OpenAI | openai | API klíč | ✅ | ✅ | ❌ | ❌ | -| Codex | openai-responses | OAuth | ✅ nuceně | ❌ | ✅ | ✅ Sazbové limity | -| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Snímky kvót | -| Kurzor | kurzor | Vlastní kontrolní součet | ✅ | ✅ | ❌ | ❌ | -| Kiro | kiro | AWS SSO OIDC | ✅ (Stream událostí) | ❌ | ✅ | ✅ Limity použití | -| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Na vyžádání | -| Qoder | openai | OAuth (základní) | ✅ | ✅ | ✅ | ⚠️ Na vyžádání | -| OpenRouter | openai | API klíč | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | claude | API klíč | ✅ | ✅ | ❌ | ❌ | -| DeepSeek | openai | API klíč | ✅ | ✅ | ❌ | ❌ | -| Groq | openai | API klíč | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | openai | API klíč | ✅ | ✅ | ❌ | ❌ | -| Mistral | openai | API klíč | ✅ | ✅ | ❌ | ❌ | -| Zmatenost | openai | API klíč | ✅ | ✅ | ❌ | ❌ | -| Společně AI | openai | Klíč API | ✅ | ✅ | ❌ | ❌ | -| Ohňostroje AI | openai | API klíč | ✅ | ✅ | ❌ | ❌ | -| Cerebras | openai | API klíč | ✅ | ✅ | ❌ | ❌ | -| Cohere | openai | API klíč | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | openai | API klíč | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Mezi zjištěné zdrojové formáty patří: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. -- "openai". -- "openai-odpovědi". -- "claude". -- "blíženci". +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | -Mezi cílové formáty patří: +All other providers (including custom compatible nodes) use the `DefaultExecutor`. -- OpenAI chat/odpovědi +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: + +- `openai` +- `openai-responses` +- `claude` +- `gemini` + +Target formats include: + +- OpenAI chat/Responses - Claude -- Gemini/Gemini-CLI/Antigravitační obálka +- Gemini/Gemini-CLI/Antigravity envelope - Kiro -- Kurzor +- Cursor -Překlady používají**OpenAI jako formát centra**— všechny konverze procházejí přes OpenAI jako prostředník:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Překlady jsou vybírány dynamicky na základě tvaru zdrojové užitečné zátěže a cílového formátu poskytovatele. +Additional processing layers in the translation pipeline: -Další vrstvy zpracování v překladovém potrubí: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Dezinfekce odezvy**– Odstraňuje nestandardní pole z odpovědí ve formátu OpenAI (streamovaných i nestreamovaných), aby byla zajištěna přísná shoda se sadou SDK --**Normalizace rolí**— Převádí `vývojář` → `systém` pro jiné cíle než OpenAI; sloučí `systém` → `uživatel` pro modely, které odmítají systémovou roli (GLM, ERNIE) -–**Extrakce značek Think**– analyzuje bloky „...“ z obsahu do pole „reasoning_content“ -–**Strukturovaný výstup**– Převádí OpenAI `response_format.json_schema` na Gemini `responseMimeType` + `responseSchema`## Supported API Endpoints +## Supported API Endpoints -| Koncový bod | Formát | Psovod | -| --------------------------------------------------- | ------------------- | -------------------------------------------------------------------- | -| `POST /v1/chat/completions` | Chat OpenAI | `src/sse/handlers/chat.ts` | -| `POST /v1/messages` | Claude Messages | Stejná obsluha (automaticky zjištěna) | -| `POST /v1/responses` | Odezvy OpenAI | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | -| `ZÍSKAT /v1/embeddings` | Seznam modelů | Cesta API | -| `POST /v1/images/generations` | Obrázky OpenAI | `open-sse/handlers/imageGeneration.ts` | -| `ZÍSKAT /v1/images/generations` | Seznam modelů | Cesta API | -| `POST /v1/providers/{provider}/chat/completions` | Chat OpenAI | Vyhrazené na poskytovatele s ověřením modelu | -| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Vyhrazené na poskytovatele s ověřením modelu | -| `POST /v1/providers/{poskytovatel}/images/generations` | Obrázky OpenAI | Vyhrazené na poskytovatele s ověřením modelu | -| `POST /v1/messages/count_tokens` | Počet tokenů Claude | Cesta API | -| `GET /v1/models` | Seznam modelů OpenAI | Cesta API (chat + vkládání + obrázek + vlastní modely) | -| `GET /api/models/catalog` | Katalog | Všechny modely seskupené podle poskytovatele + typ | -| `POST /v1beta/models/*:streamGenerateContent` | Blíženec domorodec | Cesta API | -| `GET/PUT/DELETE /api/settings/proxy` | Konfigurace proxy | Konfigurace síťového proxy | -| `POST /api/settings/proxy/test` | Připojení proxy | Koncový bod testu stavu proxy/konektivity | -| `GET/POST/DELETE /api/provider-models` | Modely poskytovatelů | Vlastní a spravované dostupné modely podporují metadata modelu poskytovatele |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -Obslužná rutina bypassu (`open-sse/utils/bypassHandler.ts`) zachycuje známé požadavky na „zahození“ od Claude CLI – zahřívací pingy, extrakce titulů a počty tokenů – a vrací**falešnou odpověď**, aniž by spotřebovával tokeny poskytovatele upstream. To se spustí pouze v případě, že `User-Agent` obsahuje `claude-cli`.## Request Logger Pipeline +## Bypass Handler -Záznamník požadavků (`open-sse/utils/requestLogger.ts`) poskytuje 7fázový kanál protokolování ladění, který je ve výchozím nastavení vypnutý, povolený pomocí `ENABLE_REQUEST_LOGS=true`:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Soubory se zapisují do `/logs//` pro každou relaci požadavku.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- Cooldown účtu poskytovatele při přechodných chybách/chybách rychlosti/autorizace -- záložní účet před neúspěšným žádostí -- Záloha kombinovaného modelu, když je vyčerpána aktuální cesta modelu/poskytovatele## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- Předběžná kontrola a obnovení s opakovaným pokusem pro poskytovatele obnovitelných zdrojů -- 401/403 opakování po pokusu o obnovení v cestě jádra## 3) Stream Safety +## 2) Token Expiry -- řadič toku s vědomím odpojení -- překladový proud s vyprázdněním konce proudu a zpracováním `[DONE]` -- záložní odhad využití, když chybí metadata využití poskytovatele## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- Objeví se chyby synchronizace, ale místní běh pokračuje -- plánovač má logiku umožňující opakování, ale periodické spouštění aktuálně standardně volá synchronizaci na jeden pokus## 5) Data Integrity +## 3) Stream Safety -- Migrace schémat SQLite a automatické upgrady při spuštění -- starší cesta ke kompatibilitě migrace JSON → SQLite## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Zdroje viditelnosti za běhu: +## 4) Cloud Sync Degradation -- protokoly konzoly z `src/sse/utils/logger.ts` -- agregáty využití na žádost v SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- čtyřfázové podrobné zachycení užitečného zatížení v SQLite (`request_detail_logs`), když `settings.detailed_logs_enabled=true` -- textový protokol o stavu požadavku v `log.txt` (nepovinné/kompatibilní) -- volitelné protokoly hlubokých požadavků/překladů pod `logs/`, když `ENABLE_REQUEST_LOGS=true` -- koncové body využití řídicího panelu (`/api/usage/*`) pro spotřebu uživatelského rozhraní +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -Podrobné zachycení datové části požadavku ukládá až čtyři fáze datové zátěže JSON na směrované volání: +## 5) Data Integrity -- nezpracovaný požadavek přijatý od klienta -- přeložená žádost skutečně odeslaná proti proudu -- odpověď poskytovatele rekonstruovaná jako JSON; streamované odpovědi jsou komprimovány do konečného shrnutí plus metadata streamu -- konečná odpověď klienta vrácená OmniRoute; streamované odpovědi jsou uloženy ve stejném kompaktním souhrnném formuláři## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- Tajný klíč JWT (`JWT_SECRET`) zajišťuje ověřování/podepisování souborů cookie relace řídicího panelu -- Počáteční zaváděcí heslo (`INITIAL_PASSWORD`) by mělo být explicitně nakonfigurováno pro zřizování při prvním spuštění -- Tajný klíč API HMAC (`API_KEY_SECRET`) zabezpečuje vygenerovaný formát lokálního klíče API -- Tajné informace poskytovatele (klíče/tokeny API) jsou uloženy v místní databázi a měly by být chráněny na úrovni souborového systému -- Koncové body synchronizace cloudu se spoléhají na sémantiku klíče API + ID počítače## Environment and Runtime Matrix +## Observability and Operational Signals -Proměnné prostředí aktivně používané kódem: +Runtime visibility sources: -- Aplikace/auth: `JWT_SECRET`, `INITIAL_PASSWORD` -- Úložiště: `DATA_DIR` -- Kompatibilní chování uzlu: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Volitelné přepsání základny úložiště (Linux/macOS, když není `DATA_DIR` nastaveno): `XDG_CONFIG_HOME` - – Bezpečnostní hash: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Protokolování: `ENABLE_REQUEST_LOGS` - – Synchronizace/cloudové URL: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` - – Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` a varianty s malými písmeny -- Příznaky funkce SOCKS5: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- Pomocníci platformy/běhu (nikoli konfigurace specifická pro aplikaci): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` a `localDb` sdílejí stejnou zásadu základního adresáře (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) se starší migrací souborů. -2. `/api/v1/route.ts` deleguje stejný tvůrce jednotného katalogu, který používá `/api/v1/models` (`src/app/api/v1/models/catalog.ts`), aby se zabránilo sémantickému posunu. -3. Pokud je povoleno, zapisovač požadavku zapisuje celé záhlaví/tělo; považovat adresář log za citlivý. -4. Chování cloudu závisí na správné dosažitelnosti koncového bodu cloudu „NEXT_PUBLIC_BASE_URL“. -5. Adresář `open-sse/` je publikován jako balíček `@omniroute/open-sse`**npm workspace**. Zdrojový kód jej importuje přes `@omniroute/open-sse/...` (vyřešeno Next.js `transpilePackages`). Cesty k souborům v tomto dokumentu stále používají název adresáře `open-sse/` kvůli konzistenci. -6. Grafy v řídicím panelu používají**Recharts**(založené na SVG) pro přístupné, interaktivní analytické vizualizace (sloupcové grafy využití modelu, tabulky rozdělení poskytovatelů s mírou úspěšnosti). -7. E2E testy používají**Playwright**(`tests/e2e/`), spouštěné přes `npm run test:e2e`. Unit testy používají**Node.js test runner**(`tests/unit/`), spouštějí se přes `npm run test:unit`. Zdrojový kód pod `src/` je**TypeScript**(`.ts`/`.tsx`); pracovní prostor `open-sse/` zůstává JavaScriptem (`.js`). -8. Stránka Nastavení je uspořádána do 5 záložek: Zabezpečení, Směrování (6 globálních strategií: fill-first, round-robin, p2c, náhodné, nejméně používané, nákladově optimalizované), Odolnost (upravitelné rychlostní limity, jistič, zásady), AI (rozpočet myšlení, systémová výzva, mezipaměť výzvy), Pokročilé (proxy).## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- Sestavení ze zdroje: `npm run build` -- Sestavení obrazu Dockeru: `docker build -t omniroute .` -- Spusťte službu a ověřte: +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: - `GET /api/settings` - `GET /api/v1/models` -- Základní adresa URL cíle CLI by měla být `http://:20128/v1`, když `PORT=20128` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/cs/docs/FEATURES.md b/docs/i18n/cs/docs/FEATURES.md index cb8205cd0f..0b538261e5 100644 --- a/docs/i18n/cs/docs/FEATURES.md +++ b/docs/i18n/cs/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Vizuální průvodce každou částí řídicího panelu OmniRoute.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Správa připojení poskytovatelů AI: poskytovatelé OAuth (Claude Code, Codex, Gemini CLI), poskytovatelé klíčů API (Groq, DeepSeek, OpenRouter) a bezplatní poskytovatelé (Qoder, Qwen, Kiro). Účty Kiro zahrnují sledování zůstatku kreditu – zbývající kredity, celkový příspěvek a datum obnovení jsou viditelné v Dashboard → Použití.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Vytvářejte komba směrování modelů se 6 strategiemi: prioritní, vážená, cyklická, náhodná, nejméně používaná a nákladově optimalizovaná. Každé kombo řetězí více modelů s automatickým nouzovým návratem a zahrnuje rychlé šablony a kontroly připravenosti.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Komplexní analýzy využití se spotřebou tokenů, odhady nákladů, teplotní mapy aktivit, týdenní distribuční grafy a rozpisy podle poskytovatelů.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Monitorování v reálném čase: doba provozuschopnosti, paměť, verze, percentily latence (p50/p95/p99), statistika mezipaměti a stavy jističe poskytovatele.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Čtyři režimy pro ladění překladů API:**Playground**(konvertor formátů),**Chat Tester**(živé požadavky),**Test Bench**(dávkové testy) a**Live Monitor**(stream v reálném čase).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Otestujte jakýkoli model přímo z palubní desky. Vyberte poskytovatele, model a koncový bod, pište výzvy pomocí editoru Monaco, streamujte odpovědi v reálném čase, rušte uprostřed streamu a zobrazujte metriky časování.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Přizpůsobitelné barevné motivy pro celý přístrojový panel. Vyberte si ze 7 přednastavených barev (korálová, modrá, červená, zelená, fialová, oranžová, azurová) nebo si vytvořte vlastní motiv výběrem libovolné šestihranné barvy. Podporuje světlý, tmavý a systémový režim.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Komplexní panel nastavení s kartami: +Comprehensive settings panel with tabs: --**Obecné**— Systémové úložiště, správa zálohování (export/import databáze) -**Vzhled**— Volič motivu (tmavý/světlý/systém), přednastavení barevných motivů a vlastní barvy, viditelnost zdravotního deníku, ovládací prvky viditelnosti položek na postranním panelu -**Zabezpečení**— Ochrana koncových bodů API, blokování vlastního poskytovatele, filtrování IP, informace o relaci -**Směrování**— Modelové aliasy, degradace úloh na pozadí -**Odolnost**- Perzistence rychlostního limitu, ladění jističe, automatické deaktivace zakázaných účtů, sledování expirace poskytovatele -**Advanced**– Přepisy konfigurace, auditní záznam konfigurace, režim degradace záložního řešení![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Konfigurace jedním kliknutím pro nástroje pro kódování AI: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor a Factory Droid. Obsahuje automatické nastavení konfigurace/resetování, profily připojení a mapování modelu.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Dashboard pro zjišťování a správu agentů CLI. Zobrazuje mřížku 14 vestavěných agentů (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) s: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Stav instalace**— Instalováno / Nenalezeno s detekcí verze -**Odznaky protokolu**— stdio, HTTP atd. -**Vlastní agenti**— Zaregistrujte jakýkoli nástroj CLI prostřednictvím formuláře (název, binární soubor, příkaz verze, spawn args) -**CLI Fingerprint Matching**– Přepínání na jednotlivé poskytovatele, aby odpovídalo nativním podpisům požadavků CLI, čímž se snižuje riziko zákazu při zachování IP adresy proxy--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Generujte obrázky, videa a hudbu z řídicího panelu. Podporuje OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open a MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Protokolování požadavků v reálném čase s filtrováním podle poskytovatele, modelu, účtu a klíče API. Zobrazuje stavové kódy, využití tokenu, latenci a podrobnosti o odpovědi.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Váš sjednocený koncový bod API s rozdělením schopností: Dokončení chatu, API odpovědí, vkládání, generování obrázků, změna pořadí, přepis zvuku, převod textu na řeč, moderování a registrované klíče rozhraní API. Integrace Cloudflare Quick Tunnel a podpora cloudového proxy pro vzdálený přístup.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -Vytvářejte, upravujte a rušte klíče API. Každý klíč může být omezen na konkrétní modely/poskytovatele s plným přístupem nebo oprávněním pouze pro čtení. Vizuální správa klíčů se sledováním využití.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Sledování administrativních akcí s filtrováním podle typu akce, aktéra, cíle, IP adresy a časového razítka. Úplná historie událostí zabezpečení.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Nativní desktopová aplikace Electron pro Windows, macOS a Linux. Spusťte OmniRoute jako samostatnou aplikaci s integrací na systémové liště, offline podporou, automatickou aktualizací a instalací jedním kliknutím. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Klíčové vlastnosti: +Key features: -- Dotazování připravenosti serveru (žádná prázdná obrazovka při studeném startu) -- Systémová lišta se správou portů -- Zásady zabezpečení obsahu -- Jednoinstanční zámek -- Automatická aktualizace při restartu -- Platformově podmíněné uživatelské rozhraní (semafory macOS, výchozí titulek Windows/Linux) -- Balení sestavení Hardened Electron – symbolicky propojené `node_modules` v samostatném balíčku jsou detekovány a odmítnuty před zabalením, čímž se zabrání závislosti běhu na sestavení (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Úplnou dokumentaci naleznete v [`electron/README.md`](../electron/README.md). +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/cs/docs/TROUBLESHOOTING.md b/docs/i18n/cs/docs/TROUBLESHOOTING.md index a6efcaa9fa..8b2b5824f2 100644 --- a/docs/i18n/cs/docs/TROUBLESHOOTING.md +++ b/docs/i18n/cs/docs/TROUBLESHOOTING.md @@ -4,69 +4,142 @@ --- -Běžné problémy a řešení pro OmniRoute.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Problém | Řešení | -| ---------------------------------------- | ------------------------------------------------------------------------------------- | --- | -| První přihlášení nefunguje | Nastavit `INITIAL_PASSWORD` v `.env` (žádné napevno zakódované výchozí nastavení) | -| Dashboard se otevírá na nesprávném portu | Nastavit `PORT=20128` a `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Žádné záznamy požadavků pod `logs/` | Nastavte `ENABLE_REQUEST_LOGS=true` | -| EACCES: povolení odepřeno | Nastavte `DATA_DIR=/cesta/k/zapisovatelnému/adresáři` tak, aby přepsal `~/.omniroute` | -| Strategie směrování se neukládá | Aktualizace na v1.4.11+ (oprava schématu Zod pro trvalost nastavení) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Příčina:**Kvóta poskytovatele je vyčerpána. +**Cause:** Provider quota exhausted. -**Oprava:** +**Fix:** -1. Zkontrolujte sledování kvót na řídicím panelu -2. Použijte kombinaci se záložními úrovněmi -3. Přejděte na levnější/bezplatnou úroveň### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Příčina:**Vyčerpaná kvóta předplatného. +### Rate Limiting -**Oprava:** +**Cause:** Subscription quota exhausted. -– Přidejte záložní: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +**Fix:** -- Použijte GLM/MiniMax jako levnou zálohu### OAuth Token Expired +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -OmniRoute automaticky obnovuje tokeny. Pokud problémy přetrvávají: +### OAuth Token Expired -1. Ovládací panel → Poskytovatel → Znovu připojit -2. Odstraňte a znovu přidejte připojení poskytovatele--- +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Ověřte, že `BASE_URL` odkazuje na vaši spuštěnou instanci (např. `http://localhost:20128`) -2. Ověřte, že `CLOUD_URL` odkazuje na váš koncový bod cloudu (např. `https://omniroute.dev`) -3. Udržujte hodnoty `NEXT_PUBLIC_*` zarovnané s hodnotami na straně serveru### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Příznak:**`Neočekávaný token 'd'...` na koncovém bodu cloudu pro nestreamovaná volání. +### Cloud `stream=false` Returns 500 -**Příčina:**Upstream vrací užitečné zatížení SSE, zatímco klient očekává JSON. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Řešení:**Pro přímá cloudová volání použijte `stream=true`. Místní běhové prostředí zahrnuje záložní SSE→JSON.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Vytvořte nový klíč z místního řídicího panelu (`/api/keys`) -2. Spusťte synchronizaci s cloudem: Povolte cloud → Synchronizovat nyní -3. Staré/nesynchronizované klíče mohou v cloudu stále vracet „401“.--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Zkontrolujte pole runtime: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. Pro přenosný režim: použijte cíl obrazu `runner-cli` (přibalená rozhraní CLI) -3. Pro režim připojení hostitele: nastavte `CLI_EXTRA_PATHS` a připojte adresář hostitele bin jako pouze pro čtení -4. Pokud `installed=true` a `runnable=false`: binární soubor byl nalezen, ale neprošel zdravotní kontrolou### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -80,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Zkontrolujte statistiky využití v Dashboard → Usage -2. Přepněte primární model na GLM/MiniMax -3. Pro nekritické úkoly používejte bezplatnou vrstvu (Gemini CLI, Qoder). -4. Nastavte rozpočty nákladů na klíč API: Dashboard → API Keys → Budget--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -V souboru `.env` nastavte `ENABLE_REQUEST_LOGS=true`. Protokoly se zobrazují v adresáři `logs/`.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -101,106 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Hlavní stav: `${DATA_DIR}/storage.sqlite` (poskytovatelé, komba, aliasy, klíče, nastavení) -- Použití: SQLite tabulky v `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + volitelné `${DATA_DIR}/log.txt` a `${DATA_DIR}/call_logs/` -- Protokoly požadavků: `/logs/...` (když `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Když je jistič poskytovatele OTEVŘENÝ, požadavky jsou blokovány, dokud nevyprší cooldown. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Oprava:** +**Fix:** -1. Přejděte na**Hlavní panel → Nastavení → Odolnost** -2. Zkontrolujte kartu jističe pro dotčeného poskytovatele -3. Kliknutím na**Resetovat vše**vymažete všechny jističe nebo počkejte, až vyprší cooldown -4. Před resetováním ověřte, zda je poskytovatel skutečně dostupný### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Pokud poskytovatel opakovaně přejde do stavu OTEVŘENO: +### Provider keeps tripping the circuit breaker -1. Zkontrolujte**Dashboard → Health → Provider Health**pro vzor selhání -2. Přejděte na**Nastavení → Odolnost → Profily poskytovatelů**a zvyšte práh selhání -3. Zkontrolujte, zda poskytovatel nezměnil limity API nebo vyžaduje opětovné ověření -4. Zkontrolujte telemetrii latence – vysoká latence může způsobit selhání na základě časového limitu--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Ujistěte se, že používáte správnou předponu: `deepgram/nova-3` nebo `assemblyai/best` - – Ověřte, že je poskytovatel připojen v**Dashboard → Providers**### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Zkontrolujte podporované zvukové formáty: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- Ověřte, zda je velikost souboru v rámci limitů poskytovatele (obvykle < 25 MB) -- Zkontrolujte platnost klíče API poskytovatele na kartě poskytovatele--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -K ladění problémů s překladem formátu použijte**Dashboard → Translator**: +Use **Dashboard → Translator** to debug format translation issues: -| Režim | Kdy použít | -| -------------------- | ----------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Hřiště** | Porovnejte vstupní/výstupní formáty vedle sebe — vložte neúspěšný požadavek, abyste viděli, jak se překládá | -| **Chat Tester** | Odesílejte živé zprávy a kontrolujte celý obsah požadavku/odpovědi včetně záhlaví | -| **Zkušební stolice** | Spusťte dávkové testy napříč kombinacemi formátů, abyste zjistili, které překlady jsou poškozené | -| **Živý monitor** | Sledujte tok požadavků v reálném čase, abyste zachytili občasné problémy s překladem | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Značky myšlení se nezobrazují**— Zkontrolujte, zda cílový poskytovatel podporuje myšlení a nastavení rozpočtu na myšlení -**Přerušení volání nástroje**— Některé překlady formátů mohou odstranit nepodporovaná pole; ověřit v režimu Playground -**Chybí systémová výzva**– Claude a Gemini zacházejí s výzvami systému odlišně; zkontrolovat překladový výstup -–**SDK vrací surový řetězec místo objektu**– Opraveno ve verzi 1.1.0: sanitizér odpovědi nyní odstraňuje nestandardní pole (`x_groq`, `usage_breakdown` atd.), která způsobují selhání ověření OpenAI SDK Pydantic -**GLM/ERNIE odmítá `systémovou` roli**— Opraveno ve verzi 1.1.0: normalizátor rolí automaticky spojuje systémové zprávy do uživatelských zpráv pro nekompatibilní modely +### Common format issues -- Role**`vývojáře` nebyla rozpoznána**— Opraveno ve verzi 1.1.0: automaticky převedeno na `systém` pro poskytovatele mimo OpenAI -**`json_schema` nefunguje s Gemini**– Opraveno ve verzi 1.1.0: `response_format` je nyní převeden na Gemini `responseMimeType` + `responseSchema`--- +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -– Automatický limit sazby se vztahuje pouze na poskytovatele klíčů API (nikoli OAuth/předplatné) +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -- Ověřte, zda je v**Nastavení → Odolnost → Profily poskytovatelů**povolen automatický limit rychlosti -- Zkontrolujte, zda poskytovatel vrací stavové kódy `429` nebo záhlaví `Retry-After`### Tuning exponential backoff +### Tuning exponential backoff -Profily poskytovatelů podporují tato nastavení: +Provider profiles support these settings: --**Základní zpoždění**— Počáteční doba čekání po prvním selhání (výchozí: 1s) -–**Max. zpoždění**– Maximální doba čekání (výchozí: 30 s) -**Multiplikátor**– o kolik se má prodloužit zpoždění při po sobě jdoucím selhání (výchozí: 2x)### Anti-thundering herd +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) -Když mnoho souběžných požadavků zasáhne poskytovatele s omezenou rychlostí, OmniRoute použije mutex + automatické omezování rychlosti k serializaci požadavků a prevenci kaskádových selhání. To je automatické pro poskytovatele klíčů API.--- +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Někteří uživatelé OmniRoute umístí bránu před RAG nebo zásobníky agentů. V těchto nastaveních je běžné vidět podivný vzorec: OmniRoute vypadá zdravě (poskytovatelé jsou v pořádku, směrovací profily jsou v pořádku, žádná upozornění na omezení rychlosti), ale konečná odpověď je stále špatná. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -V praxi tyto incidenty obvykle pocházejí z navazujícího potrubí RAG, nikoli ze samotné brány. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Pokud chcete sdílený slovník pro popis těchto selhání, můžete použít WFGY ProblemMap, externí textový zdroj licence MIT, který definuje šestnáct opakujících se vzorců selhání RAG / LLM. Na vysoké úrovni pokrývá: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- posun vyhledávání a porušené hranice kontextu -- prázdné nebo zastaralé indexy a vektorová úložiště -- vkládání versus sémantický nesoulad -- rychlé sestavení a problémy s kontextovým oknem -- logický kolaps a příliš sebevědomé odpovědi -- selhání koordinace dlouhých řetězců a agentů -- multiagentní paměť a posun rolí -- problémy s nasazením a objednáním bootstrapu +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -Myšlenka je jednoduchá: +The idea is simple: -1. Když prozkoumáte špatnou odpověď, zachyťte: - - uživatelský úkol a požadavek - - kombinace trasy nebo poskytovatele v OmniRoute - - jakýkoli kontext RAG použitý po proudu (načtené dokumenty, volání nástrojů atd.) -2. Namapujte incident na jedno nebo dvě čísla WFGY ProblemMap (`č.1` … `č.16`). -3. Uložte číslo na svůj vlastní řídicí panel, runbook nebo sledovač incidentů vedle protokolů OmniRoute. -4. Použijte příslušnou stránku WFGY k rozhodnutí, zda potřebujete změnit strategii zásobníku RAG, retrieveru nebo směrování. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -Celý text a konkrétní recepty jsou k dispozici zde (licence MIT, pouze text): +Full text and concrete recipes live here (MIT license, text only): -[SOUBOR WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) +[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -Tuto sekci můžete ignorovat, pokud za OmniRoute nespouštíte RAG nebo agenty.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? -–**Problémy s GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Architecture**: Interní podrobnosti viz [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) -**Reference API**: Všechny koncové body viz [`docs/API_REFERENCE.md`](API_REFERENCE.md) -**Health Dashboard**: Zkontrolujte**Dashboard → Health**pro stav systému v reálném čase -**Translator**: K ladění problémů s formátem použijte**Dashboard → Translator** +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/cs/llm.txt b/docs/i18n/cs/llm.txt new file mode 100644 index 0000000000..79760c56fd --- /dev/null +++ b/docs/i18n/cs/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Čeština) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Přehled + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Bezpečnost +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/da/README.md b/docs/i18n/da/README.md index 07c3e3ee50..06a0559457 100644 --- a/docs/i18n/da/README.md +++ b/docs/i18n/da/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Din universelle API-proxy — ét slutpunkt, 60+ udbydere, ingen nedetid. Nu med**MCP Server (25 værktøjer)**,**A2A Protocol**,**Memory/Skills Systems**&**Electron Desktop App**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Chatafslutninger • Indlejringer • Billedgenerering • Video • Musik • Lyd • Genrangering •**Websøgning**• MCP-server • A2A-protokol • 100 % TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Din universelle API-proxy — ét slutpunkt, 60+ udbydere, ingen nedetid. Nu me [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Hjemmeside](https://omniroute.online) • [🚀 Lynstart](#-hurtig-start) • [💡 Funktioner](#-nøglefunktioner) • [📖 Docs](#-dokumentation) • [💰 Priser](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Tilgængelig på:**🇺🇸 [engelsk](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Tysk](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [English](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesien](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [filippinsk](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,28 +60,30 @@ _Din universelle API-proxy — ét slutpunkt, 60+ udbydere, ingen nedetid. Nu me ## 📸 Dashboard Preview - -Klik for at se skærmbilleder af dashboard +
+Click to see dashboard screenshots -| Side | Skærmbillede | -| ----------------- | -------------------------------------------------- | ---------- | -| **Udbydere** | ![Providers](docs/screenshots/01-providers.png) | -| **Komboer** | ![Combos](docs/screenshots/02-combos.png) | -| **Analyse** | ![Analytics](docs/screenshots/03-analytics.png) | -| **Sundhed** | ![Health](docs/screenshots/04-health.png) | -| **Oversætter** | ![Oversætter](docs/screenshots/05-translator.png) | -| **Indstillinger** | ![Indstillinger](docs/screenshots/06-settings.png) | -| **CLI-værktøjer** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | -| **Brugslogfiler** | ![Usage](docs/screenshots/08-usage.png) | -| **Endpunkter** | ![Endpoints](docs/screenshots/09-endpoint.png) |
| +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | + + --- ### 🤖 Free AI Provider for your favorite coding agents -_Tilslut ethvert AI-drevet IDE- eller CLI-værktøj gennem OmniRoute - gratis API-gateway til ubegrænset kodning._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ - + @@ -125,481 +134,555 @@ _Tilslut ethvert AI-drevet IDE- eller CLI-værktøj gennem OmniRoute - gratis AP Codex CLI
Codex CLI
- ⭐ 60,8K + ⭐ 60.8K
@@ -88,28 +97,28 @@ _Tilslut ethvert AI-drevet IDE- eller CLI-værktøj gennem OmniRoute - gratis AP NanoBot
NanoBot

- ⭐ 20,9K + ⭐ 20.9K
PicoClaw
PicoClaw

- ⭐ 14,6K + ⭐ 14.6K
ZeroClaw
ZeroClaw

- ⭐ 9,9K + ⭐ 9.9K
IronClaw
IronClaw

- ⭐ 2,1K + ⭐ 2.1K
Claude Code
Claude Code

- ⭐ 67,3K + ⭐ 67.3K
Gemini CLI
Gemini CLI

- ⭐ 94,7K + ⭐ 94.7K
- Kilokode
- Kilokode + Kilo Code
+ Kilo Code

- ⭐ 15,5K + ⭐ 15.5K
-📡 Alle agenter forbinder via http://localhost:20128/v1 eller http://cloud.omniroute.online/v1 - én konfiguration, ubegrænset modeller og kvote--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Stop med at spilde penge og nå grænser:** +**Stop wasting money and hitting limits:** -- Abonnementskvoten udløber ubrugt hver måned -- Satsgrænser stopper dig med at midtkode -- Dyre API'er ($20-50/måned pr. udbyder) -- Manuel skift mellem udbydere +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute løser dette:** +**OmniRoute solves this:** -- ✅**Maksimer abonnementer**- Spor kvote, brug hver bit før nulstilling -- ✅**Automatisk fallback**- Abonnement → API-nøgle → Billig → Gratis, ingen nedetid -- ✅**Multi-konto**- Round-robin mellem konti pr. udbyder -- ✅**Universal**- Virker med Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, ethvert CLI-værktøj--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Tilmeld dig vores fællesskab!**[WhatsApp-gruppe](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Få hjælp, del tips, og hold dig opdateret. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Websted**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Problemer**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Fællesskabsgruppe](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Bidrager**: Se [CONTRIBUTING.md](CONTRIBUTING.md), åbn en PR, eller vælg et "godt første nummer" -**Originalt projekt**: [9router af decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Når du åbner et problem, skal du køre kommandoen systeminfo og vedhæfte den genererede fil:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Dette genererer en `system-info.txt` med din Node.js-version, OmniRoute-version, OS-detaljer, installerede CLI-værktøjer (qoder, gemini, claude, codex, antigravity, droid osv.), Docker/PM2-status og systempakker - alt hvad vi har brug for for hurtigt at reproducere dit problem. Vedhæft filen direkte til dit GitHub-problem.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Alle udviklere, der bruger AI-værktøjer, står over for disse problemer dagligt.**OmniRoute blev bygget til at løse dem alle - fra omkostningsoverskridelser til regionale blokke, fra ødelagte OAuth-flows til protokoloperationer og observerbarhed i virksomheden. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. - -💸 1. "Jeg betaler for et dyrt abonnement, men bliver stadig afbrudt af grænser" +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Udviklere betaler $20-200/måned for Claude Pro, Codex Pro eller GitHub Copilot. Selv ved betaling har kvoten et loft - 5 timers brug, ugentlige grænser eller satsgrænser pr. minut. Mid-coding session, udbyderen holder op med at svare, og udvikleren mister flow og produktivitet. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Sådan løser OmniRoute det:** +**How OmniRoute solves it:** --**Smart 4-Tier Fallback**— Hvis abonnementskvoten løber ud, omdirigeres automatisk til API Key → Billig → Gratis uden manuel indgriben --**Sporing af udbydergrænser**— Cachelagrede kvote-øjebliksbilleder opdateres på en server-sideplan (standard `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) med manuel opdatering tilgængelig i brugergrænsefladen --**Multi-Account Support**- Flere konti pr. udbyder med automatisk round-robin - når den ene løber tør, skifter til den næste --**Brugerdefinerede kombinationer**— Tilpasselige fallback-kæder med 9 balanceringsstrategier (prioritet, vægtet, fill-first, round-robin, P2C, tilfældig, mindst brugt, omkostningsoptimeret, strengt tilfældig) --**Codex Business Quotas**— Business/Team Workspace kvoteovervågning direkte i dashboardet
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard - -🔌 2. "Jeg skal bruge flere udbydere, men hver har en anden API" + -OpenAI bruger et format, Claude (Antropisk) bruger et andet, Gemini endnu et andet. Hvis en udvikler ønsker at teste modeller fra forskellige udbydere eller fallback mellem dem, skal de omkonfigurere SDK'er, ændre slutpunkter, håndtere inkompatible formater. Tilpassede udbydere (FriendLI, NIM) har ikke-standardmodelslutpunkter. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Sådan løser OmniRoute det:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Unified Endpoint**— En enkelt `http://localhost:20128/v1` fungerer som proxy for alle 60+ udbydere --**Formatoversættelse**— Automatisk og gennemsigtig: OpenAI ↔ Claude ↔ Gemini ↔ Responses API --**Response Sanitization**- Fjerner ikke-standardfelter (`x_groq`, `usage_breakdown`, `service_tier`), der bryder OpenAI SDK v1.83+ --**Rollenormalisering**— Konverterer `udvikler` → `system` for ikke-OpenAI-udbydere; `system` → `bruger` til GLM/ERNIE --**Think Tag Extraction**— Udtrækker ""-blokke fra modeller som DeepSeek R1 til standardiseret "reasoning_content" --**Structured Output for Gemini**— `json_schema` → `responseMimeType`/`responseSchema` automatisk konvertering --**`stream` er standard til "false"**- Justerer med OpenAI-specifikationer, undgår uventede SSE i Python/Rust/Go SDK'er
+**How OmniRoute solves it:** - -🌐 3. "Min AI-udbyder blokerer mit område/land" +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Udbydere som OpenAI/Codex blokerer adgang fra visse geografiske områder. Brugere får fejl som "unsupported_country_region_territory" under OAuth- og API-forbindelser. Dette er især frustrerende for udviklere fra udviklingslande. + -**Sådan løser OmniRoute det:** +
+🌐 3. "My AI provider blocks my region/country" --**3-Level Proxy Config**— Konfigurerbar proxy på 3 niveauer: global (al trafik), pr. udbyder (kun én udbyder) og pr. forbindelse/nøgle --**Farvekodede proxy-badges**— Visuelle indikatorer: 🟢 global proxy, 🟡 udbyder proxy, 🔵 forbindelsesproxy, viser altid IP'en --**OAuth-tokenudveksling gennem proxy**- OAuth-flowet går også gennem proxyen og løser "unsupported_country_region_territory". --**Forbindelsestest via proxy**— Forbindelsestest bruger den konfigurerede proxy (ikke mere direkte omgåelse) --**SOCKS5-understøttelse**— Fuld SOCKS5-proxy-understøttelse til udgående routing --**TLS Fingerprint Spoofing**— Browserlignende TLS-fingeraftryk via 'wreq-js' for at omgå botdetektion --**🔏 Matching af CLI-fingeraftryk**— Omarrangerer overskrifter og kropsfelter, så de matcher native CLI-binære signaturer, hvilket drastisk reducerer risikoen for kontoflaggning. Proxy-IP'en bevares - du får både stealth**og**IP-maskering samtidigt
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. - -🆓 4. "Jeg vil bruge AI til kodning, men jeg har ingen penge" +**How OmniRoute solves it:** -Ikke alle kan betale $20-200/måned for AI-abonnementer. Studerende, udviklere fra vækstlande, hobbyfolk og freelancere har brug for adgang til kvalitetsmodeller uden omkostninger. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Sådan løser OmniRoute det:** + --**Free Tier Providers Indbygget**— Indbygget understøttelse af 100 % gratis udbydere: Qoder (5 ubegrænsede modeller via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited-modeller:-r-modeller:-r qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID gratis), Gemini CLI (180K tokens/måned gratis) --**Ollama Cloud**— Cloud-hostede Ollama-modeller på `api.ollama.com` med gratis "Light usage"-niveau; brug `ollamacloud/` præfiks --**Kun gratis kombinationer**— Kæde `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/måned uden nedetid --**NVIDIA NIM Free Access**— ~40 RPM dev-forever gratis adgang til 70+ modeller på build.nvidia.com (overgang fra kreditter til rene hastighedsgrænser) --**Cost Optimized Strategy**— Routingstrategi, der automatisk vælger den billigste tilgængelige udbyder +
+🆓 4. "I want to use AI for coding but I have no money" - -🔒 5. "Jeg skal beskytte min AI-gateway mod uautoriseret adgang" +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -Når en AI-gateway eksponeres for netværket (LAN, VPS, Docker), kan enhver med adressen forbruge udviklerens tokens/kvote. Uden beskyttelse er API'er sårbare over for misbrug, hurtig injektion og misbrug. +**How OmniRoute solves it:** -**Sådan løser OmniRoute det:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**API Key Management**— Generering, rotation og scoping pr. udbyder med en dedikeret `/dashboard/api-manager`-side --**Tilladelser på modelniveau**— Begræns API-nøgler til specifikke modeller ('openai/*', jokertegnsmønstre) med Tillad alt/Begræns-skift --**API Endpoint Protection**— Kræv en nøgle til `/v1/modeller` og bloker specifikke udbydere fra listen --**Auth Guard + CSRF Protection**— Alle dashboard-ruter beskyttet med 'withAuth' middleware + CSRF-tokens --**Rate Limiter**— Per-IP hastighedsbegrænsning med konfigurerbare vinduer --**IP-filtrering**— Tilladelsesliste/blokeringsliste til adgangskontrol --**Prompt Injection Guard**— Sanering mod ondsindede promptmønstre --**AES-256-GCM-kryptering**— Legitimationsoplysninger krypteret i hvile
+ - -🛑 6. "Min udbyder gik ned, og jeg mistede mit kodningsflow" +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -AI-udbydere kan blive ustabile, returnere 5xx-fejl eller ramme midlertidige hastighedsgrænser. Hvis en udvikler afhænger af en enkelt udbyder, bliver de afbrudt. Uden strømafbrydere kan gentagne genforsøg crashe programmet. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Sådan løser OmniRoute det:** +**How OmniRoute solves it:** --**Circuit Breaker pr. model**— Automatisk åbning/lukning med konfigurerbare tærskler og nedkøling (Lukket/Åben/Halv-Åben), omfang pr. model for at undgå kaskadeblokke --**Eksponentiel backoff**— Progressive forsinkelser af genforsøg --**Anti-tordenbesætning**— Mutex + semaforbeskyttelse mod samtidige genforsøgsstorme --**Combo Fallback Chains**— Hvis den primære udbyder fejler, falder den automatisk gennem kæden uden indgriben --**Combo Circuit Breaker**- Deaktiverer automatisk fejlende udbydere i en kombinationskæde --**Health Dashboard**— Oppetidsovervågning, strømafbrydertilstande, lockouts, cachestatistik, p50/p95/p99 latency
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest - -🔧 7. "Konfiguration af hvert AI-værktøj er trættende og gentagende" + -Udviklere bruger Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Hvert værktøj har brug for en anden konfiguration (API-endepunkt, nøgle, model). At omkonfigurere, når du skifter udbyder eller model, er spild af tid. +
+🛑 6. "My provider went down and I lost my coding flow" -**Sådan løser OmniRoute det:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**CLI Tools Dashboard**— Dedikeret side med et-klik opsætning til Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline --**GitHub Copilot Config Generator**— Genererer `chatLanguageModels.json` til VS-kode med bulk modelvalg --**Onboarding Wizard**— Guidet 4-trins opsætning for førstegangsbrugere --**Et slutpunkt, alle modeller**— Konfigurer `http://localhost:20128/v1` én gang, få adgang til 60+ udbydere
+**How OmniRoute solves it:** - -🔑 8. "Administration af OAuth-tokens fra flere udbydere er et helvede" +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot - alle bruger OAuth 2.0 med udløbende tokens. Udviklere skal genautentificere konstant, håndtere `client_secret is missing`, `redirect_uri_mismatch` og fejl på fjernservere. OAuth på LAN/VPS er særligt problematisk. + -**Sådan løser OmniRoute det:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Automatisk tokenopdatering**— OAuth-tokens opdateres i baggrunden før udløb --**OAuth 2.0 (PKCE) Indbygget**— Automatisk flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder --**Multi-Account OAuth**— Flere konti pr. udbyder via JWT/ID-tokenudtrækning --**OAuth LAN/Remote Fix**— Privat IP-detektion for `redirect_uri` + manuel URL-tilstand for fjernservere --**OAuth Behind Nginx**— Bruger `window.location.origin` til omvendt proxy-kompatibilitet --**Remote OAuth Guide**— Trin-for-trin guide til Google Cloud-legitimationsoplysninger på VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. - -📊 9. "Jeg ved ikke, hvor meget jeg bruger eller hvor" +**How OmniRoute solves it:** -Udviklere bruger flere betalte udbydere, men har ikke noget samlet syn på udgifter. Hver udbyder har sit eget faktureringsdashboard, men der er ingen konsolideret visning. Uventede omkostninger kan hobe sig op. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Sådan løser OmniRoute det:** + --**Dashboard for omkostningsanalyse**— omkostningssporing pr. token og budgetstyring pr. udbyder --**Budgetgrænser pr. niveau**— Udgiftsloft pr. niveau, der udløser automatisk fallback --**Priskonfiguration pr. model**— Konfigurerbare priser pr. model --**Brugsstatistik pr. API-nøgle**— Antal anmodninger og sidst anvendte tidsstempel pr. nøgle --**Analytics Dashboard**— Statiske kort, modelbrugsdiagram, udbydertabel med succesrater og latens +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" - -🐛 10. "Jeg kan ikke diagnosticere fejl og problemer i AI-kald" +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Når et opkald mislykkes, ved udvikleren ikke, om det var en takstgrænse, udløbet token, forkert format eller udbyderfejl. Fragmenterede logfiler på tværs af forskellige terminaler. Uden observerbarhed er fejlfinding trial-and-error. +**How OmniRoute solves it:** -**Sådan løser OmniRoute det:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Unified Logs Dashboard**— 4 faner: Request Logs, Proxy Logs, Audit Logs, Console --**Console Log Viewer**— Realtidsterminal-fremviser med farvekodede niveauer, automatisk rulning, søg, filtrer --**SQLite Proxy Logs**— Vedvarende logfiler, der overlever servergenstarter --**Oversætterlegeplads**— 4 fejlfindingstilstande: Legeplads (formatoversættelse), Chattester (rundtur), Testbænk (batch), Live Monitor (realtid) --**Request Telemetri**— p50/p95/p99 latency + X-Request-Id-sporing --**Filbaseret logning med rotation**— Applogfiler roterer efter størrelse, opbevaringsdage og arkivantal; opkaldslog-artefakter roterer efter opbevaringsdage og filantal --**System Info Report**— `npm run system-info` genererer `system-info.txt` med dit fulde miljø (Nodeversion, OmniRoute-version, OS, CLI-værktøjer, Docker/PM2-status). Vedhæft det, når du rapporterer problemer til øjeblikkelig triage.
+ - -🏗️ 11. "Deployering og vedligeholdelse af gatewayen er kompleks" +
+📊 9. "I don't know how much I'm spending or where" -Installation, konfiguration og vedligeholdelse af en AI-proxy på tværs af forskellige miljøer (lokalt, VPS, Docker, cloud) er arbejdskrævende. Problemer som hårdkodede stier, "EACCES" på mapper, portkonflikter og cross-platform builds tilføjer friktion. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Sådan løser OmniRoute det:** +**How OmniRoute solves it:** --**npm global installation**— `npm install -g omniroute && omniroute` — færdig --**Docker Multi-Platform**— AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) --**Docker Compose Profiles**— 'base' (ingen CLI-værktøjer) og 'cli' (med Claude Code, Codex, OpenClaw) --**Electron Desktop App**— Indbygget app til Windows/macOS/Linux med systembakke, autostart, offlinetilstand --**Split-Port Mode**— API og Dashboard på separate porte til avancerede scenarier (omvendt proxy, containernetværk) --**Cloud Sync**— Konfigurer synkronisering på tværs af enheder via Cloudflare Workers --**DB Backups**— Automatisk backup, gendannelse, eksport og import af alle indstillinger med `DISABLE_SQLITE_AUTO_BACKUP` til eksternt administrerede sikkerhedskopier
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency - -🌍 12. "Grænsefladen er kun engelsk, og mit team taler ikke engelsk" + -Hold i ikke-engelsktalende lande, især i Latinamerika, Asien og Europa, kæmper med grænseflader, der kun er på engelsk. Sprogbarrierer reducerer adoption og øger konfigurationsfejl. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Sådan løser OmniRoute det:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Dashboard i18n — 30 sprog**— Alle 500+ taster oversat, inklusive arabisk, bulgarsk, dansk, tysk, spansk, finsk, fransk, hebraisk, hindi, ungarsk, indonesisk, italiensk, japansk, koreansk, malaysisk, hollandsk, norsk, polsk, portugisisk (PT/BR), rumænsk, russisk, ukrainsk, kinesisk, ukrainsk, kinesisk, kinesisk, ukrainsk, kinesisk, ukrainsk, kinesisk, ukrainsk, svensk, Vietnam, Vietnam --**RTL-understøttelse**— Højre-til-venstre-understøttelse for arabisk og hebraisk --**Multi-Language READMEs**— 30 komplette dokumentationsoversættelser --**Sprogvælger**— Globusikon i overskriften til skift i realtid
+**How OmniRoute solves it:** - -🔄 13. "Jeg har brug for mere end chat – jeg har brug for indlejringer, billeder, lyd" +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -AI er ikke bare fuldførelse af chat. Udviklere skal generere billeder, transskribere lyd, oprette indlejringer til RAG, omrangere dokumenter og moderere indhold. Hver API har et andet slutpunkt og format. + -**Sådan løser OmniRoute det:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Embeddings**— `/v1/embeddings` med 6 udbydere og 9+ modeller --**Billedgenerering**— `/v1/images/generations` med 10 udbydere og 20+ modeller (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Tekst-til-video**— `/v1/videoer/generationer` — ComfyUI (AnimateDiff, SVD) og SD WebUI --**Text-to-Music**— `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) --**Lydtransskription**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Text-to-Speech**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + eksisterende udbydere --**Moderationer**— `/v1/moderations` — Indholdssikkerhedstjek --**Reranking**— `/v1/rerank' — Reranking af dokumentrelevans --**Responses API**— Fuld `/v1/responses`-understøttelse af Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. - -🧪 14. "Jeg har ingen måde at teste og sammenligne kvalitet på tværs af modeller" +**How OmniRoute solves it:** -Udviklere vil gerne vide, hvilken model der er bedst til deres brug - kode, oversættelse, ræsonnement - men manuel sammenligning er langsom. Der findes ingen integrerede evalueringsværktøjer. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**Sådan løser OmniRoute det:** + --**LLM-evalueringer**— Gyldne sæt-test med 10 forudindlæste cases, der dækker hilsner, matematik, geografi, kodegenerering, JSON-overholdelse, oversættelse, markdown, sikkerhedsafvisning --**4 matchstrategier**— 'præcis', 'indeholder', 'regex', 'brugerdefineret' (JS-funktion) --**Translator Playground Test Bench**— Batchtest med flere input og forventede output, sammenligning på tværs af udbydere --**Chattester**— Fuld rundtur med visuel responsgengivelse --**Live Monitor**— Realtidsstream af alle anmodninger, der flyder gennem proxyen +
+🌍 12. "The interface is English-only and my team doesn't speak English" - -📈 15. "Jeg har brug for at skalere uden at miste ydeevne" +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -Efterhånden som forespørgselsvolumen vokser, genererer de samme spørgsmål duplikerede omkostninger uden cache. Uden idempotens, dublerede anmodninger om affaldsbehandling. Takstgrænser pr. udbyder skal overholdes. +**How OmniRoute solves it:** -**Sådan løser OmniRoute det:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Semantisk cache**— To-lags cache (signatur + semantisk) reducerer omkostninger og latens --**Request Idempotency**— 5s deduplikeringsvindue for identiske anmodninger --**Detektion af hastighedsgrænse**— RPM pr. udbyder, min. gap og maks. samtidig sporing --**Redigerbare hastighedsgrænser**— Konfigurerbare standardindstillinger i Indstillinger → Modstandsdygtighed med vedholdenhed --**API Key Validation Cache**— 3-lags cache til produktionsydeevne --**Health Dashboard med telemetri**— p50/p95/p99 latency, cachestatistik, oppetid
+ - -🤖 16. "Jeg vil kontrollere modeladfærd globalt" +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Udviklere, der ønsker alle svar på et bestemt sprog, med en bestemt tone, eller ønsker at begrænse ræsonnementstokens. Det er upraktisk at konfigurere dette i hvert værktøj/anmodning. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Sådan løser OmniRoute det:** +**How OmniRoute solves it:** --**System Prompt Injection**— Global prompt anvendt på alle anmodninger --**Thinking Budget Validation**— Reasoning token allocation control pr. anmodning (passthrough, auto, custom, adaptive) --**9 Routing Strategies**— Globale strategier, der bestemmer, hvordan anmodninger distribueres --**Wildcard Router**— `udbyder/*`-mønstre rutes dynamisk til enhver udbyder --**Kombo Aktiver/Deaktiver Til/fra**— Skift kombinationer direkte fra dashboardet --**Tilskiftning af udbyder**— Aktiver/deaktiver alle forbindelser for en udbyder med et enkelt klik --**Blokerede udbydere**— Ekskluder specifikke udbydere fra `/v1/models` liste
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex - -🧰 17. "Jeg har brug for MCP-værktøjer som førsteklasses produktegenskaber" + -Mange AI-gateways afslører kun MCP som en skjult implementeringsdetalje. Teams har brug for et synligt, overskueligt operationslag. +
+🧪 14. "I have no way to test and compare quality across models" -**Sådan løser OmniRoute det:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP vises på fanen dashboardnavigation og endepunktsprotokol -- Dedikeret MCP-administrationsside med proces, værktøjer, omfang og revision -- Indbygget hurtigstart til `omniroute --mcp` og klient onboarding
+**How OmniRoute solves it:** - -🧠 18. "Jeg har brug for A2A-orkestrering med synkronisering + stream opgavestier" +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Agentarbejdsgange kræver både direkte svar og langvarig streamet udførelse med livscykluskontrol. + -**Sådan løser OmniRoute det:** +
+📈 15. "I need to scale without losing performance" -- A2A JSON-RPC-slutpunkt ('POST /a2a') med 'message/send' og 'message/stream' -- SSE-streaming med udbredelse af terminaltilstand -- Opgavelivscyklus API'er for "opgaver/hent" og "opgaver/annuller".
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. - -🛰️ 19. "Jeg har brug for ægte MCP-processundhed, ikke gættet status" +**How OmniRoute solves it:** -Operationelle teams skal vide, om MCP faktisk er i live, ikke kun om en API er tilgængelig. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**Sådan løser OmniRoute det:** + -- Runtime-hjerteslagsfil med PID, tidsstempler, transport, værktøjstælling og omfangstilstand -- MCP status API, der kombinerer hjerteslag + seneste aktivitet -- UI-statuskort til proces/oppetid/hjerteslagsfriskhed +
+🤖 16. "I want to control model behavior globally" - -📋 20. "Jeg har brug for auditable MCP-værktøjsudførelse" +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Når værktøjer muterer konfiguration eller udløser ops-handlinger, har teams brug for retsmedicinsk sporbarhed. +**How OmniRoute solves it:** -**Sådan løser OmniRoute det:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- SQLite-støttet revisionslogning for MCP-værktøjsopkald -- Filtrerer efter værktøj, succes/fiasko, API-nøgle og paginering -- Dashboard revisionstabel + statistik slutpunkter til automatisering
+ - -🔐 21. "Jeg har brug for scoped MCP-tilladelser pr. integration" +
+🧰 17. "I need MCP tools as first-class product capabilities" -Forskellige klienter bør have mindst privilegeret adgang til værktøjskategorier. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Sådan løser OmniRoute det:** +**How OmniRoute solves it:** -- 10 granulære MCP-skoper til kontrolleret værktøjsadgang -- Håndhævelse af omfang og synlighed i MCP management UI -- Sikker standardstilling for operationelt værktøj
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding - -⚙️ 22. "Jeg har brug for operationelle kontroller uden omfordeling" + -Teams har brug for hurtige runtime-ændringer under hændelser eller omkostningsbegivenheder. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**Sådan løser OmniRoute det:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Skift kombinationsaktivering direkte fra MCP-dashboard -- Anvend modstandsdygtighedsprofiler fra foruddefinerede politikpakker -- Nulstil strømafbrydertilstand fra det samme betjeningspanel
+**How OmniRoute solves it:** - -🔄 23. "Jeg har brug for live A2A opgave livscyklus synlighed og annullering" +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Uden livscyklussynlighed bliver opgavehændelser svære at triage. + -**Sådan løser OmniRoute det:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Opgaveliste/filtrering efter tilstand/færdighed med paginering -- Drill-down på opgavemetadata, hændelser og artefakter -- Slutpunkt for annullering af opgave og UI-handling med bekræftelse
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. - -🌊 24. "Jeg har brug for aktive stream-metrics for A2A-indlæsning" +**How OmniRoute solves it:** -Streaming-arbejdsgange kræver operationel indsigt i samtidighed og live-forbindelser. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**Sådan løser OmniRoute det:** + -- Aktive stream-tællere integreret i A2A-status -- Tidsstempel for sidste opgave og tæller pr. stat -- A2A dashboard-kort til operationsovervågning i realtid +
+📋 20. "I need auditable MCP tool execution" - -🪪 25. "Jeg har brug for standardagentopdagelse til klienter" +When tools mutate config or trigger ops actions, teams need forensic traceability. -Eksterne klienter og orkestratorer har brug for maskinlæsbare metadata til onboarding. +**How OmniRoute solves it:** -**Sådan løser OmniRoute det:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- Agentkort afsløret på `/.well-known/agent.json` -- Evner og færdigheder vist i ledelsens brugergrænseflade -- A2A status API inkluderer opdagelsesmetadata til automatisering
+ - -🧭 26. "Jeg har brug for protokolsynlighed i produktets UX" +
+🔐 21. "I need scoped MCP permissions per integration" -Hvis brugere ikke kan opdage protokoloverflader, falder kvaliteten af adoption og support. +Different clients should have least-privilege access to tool categories. -**Sådan løser OmniRoute det:** +**How OmniRoute solves it:** -- Konsolideret**Endpoints**-side med faner til Proxy, MCP, A2A og API Endpoints -- Inline service status skifter (Online/Offline) for MCP og A2A -- Links fra oversigt til dedikerede administrationsfaner
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling - -🧪 27. "Jeg har brug for end-to-end protokolvalidering med rigtige klienter" + -Mock-tests er ikke nok til at validere protokolkompatibilitet før frigivelse. +
+⚙️ 22. "I need operational controls without redeploying" -**Sådan løser OmniRoute det:** +Teams need quick runtime changes during incidents or cost events. -- E2E-pakke, der starter app og bruger ægte MCP SDK-klienttransport -- A2A klient tester for opdagelse, send, stream, hent og annuller flows -- Krydstjek påstande mod MCP-revision og A2A-opgaver API'er
+**How OmniRoute solves it:** - -📡 28. "Jeg har brug for samlet observerbarhed på tværs af alle grænseflader" +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -Opdeling af observerbarhed efter protokol skaber blinde pletter og længere MTTR. + -**Sådan løser OmniRoute det:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Samlede dashboards/logfiler/analyse i ét produkt -- Health + audit + request telemetri på tværs af OpenAI, MCP og A2A lag -- Operationelle API'er til status og automatisering
+Without lifecycle visibility, task incidents become hard to triage. - -💼 29. "Jeg har brug for én runtime til proxy + værktøjer + agentorkestrering" +**How OmniRoute solves it:** -At køre mange separate tjenester øger driftsomkostninger og fejltilstande. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**Sådan løser OmniRoute det:** + -- OpenAI-kompatibel proxy, MCP-server og A2A-server i én stak -- Delt godkendelse, robusthed, datalager og observerbarhed -- Ensartet politikmodel på tværs af alle interaktionsflader +
+🌊 24. "I need active stream metrics for A2A load" - -🚀 30. "Jeg skal sende agentiske arbejdsgange uden limkodesprawl" +Streaming workflows require operational insight into concurrency and live connections. -Hold mister hastighed, når de sammensætter flere ad-hoc-tjenester og scripts. +**How OmniRoute solves it:** -**Sådan løser OmniRoute det:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Ensartet slutpunktsstrategi for kunder og agenter -- Indbygget protokolstyring UI'er og røgvalideringsstier -- Produktionsklare fundamenter (sikkerhed, logning, robusthed, backup)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Playbook A: Maksimer betalt abonnement + billig backup**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -607,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Playbook B: Kodningsstak uden omkostninger**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Playbook C: 24/7 altid aktiv reservekæde**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -630,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Playbook D: Agent ops med MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Konfigurer AI-kodning på få minutter til**$0/måned**. Tilslut disse gratis konti, og brug den indbyggede**Free Stack**-kombination. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Trin | Handling | Udbydere ulåst | -| ---- | -------------------------------------------------- | -------------------------------------------------------------------------- | -| 1 | Tilslut**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**ubegrænset**| -| 2 | Tilslut**Qoder**(Google OAuth) | kimi-k2-tænkning, qwen3-coder-plus, deepseek-r1... —**ubegrænset**| -| 3 | Tilslut**Qwen**(enhedskode) | qwen3-coder-plus, qwen3-coder-flash... —**ubegrænset**| -| 4 | Tilslut**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180K/md gratis**| -| 5 | `/dashboard/combos` →**Gratis stak ($0)**skabelon | Round-robin alle gratis udbydere automatisk | +| Step | Action | Providers Unlocked | +| ---- | -------------------------------------------------- | ------------------------------------------------------------------ | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Peg enhver IDE/CLI til:**`http://localhost:20128/v1` · API-nøgle: `any-string` · Udført. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Valgfri ekstra dækning (også gratis):**Groq API-nøgle (30 RPM gratis), NVIDIA NIM (40 RPM gratis, 70+ modeller), Cerebras (1M tok/dag), LongCat API-nøgle (50M tokens/dag!), Cloudflare Workers AI (10K Neurons/day, 50+ modeller).## Kom hurtigt i gang +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Kom hurtigt i gang ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **pnpm-brugere:**Kør `pnpm approve-builds -g` efter installation for at aktivere native build-scripts, der kræves af `better-sqlite3` og `@swc/core`: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > > ```bash -> pnpm installer -g omniroute -> pnpm approve-builds -g # Vælg alle pakker → godkend +> pnpm install -g omniroute +> pnpm approve-builds -g # Select all packages → approve > omniroute > ``` -Dashboard åbner på `http://localhost:20128` og API-base-URL er `http://localhost:20128/v1`. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Kommando | Beskrivelse | +| Command | Description | | ----------------------- | ----------------------------------------------------------- | -| `omniroute` | Start server (`PORT=20128`, API og dashboard på samme port) | -| `omniroute --port 3000` | Indstil kanonisk/API-port til 3000 | -| `omniroute --mcp` | Start MCP-server (stdio-transport) | -| `omniroute --no-open` | Åbn ikke browseren automatisk | -| `omniroute --hjælp` | Vis hjælp | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Valgfri split-port-tilstand:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -Til de fleste implementeringer behøver du kun: +For most deployments, you only need: -| Variabel | Standard | Formål | -| -------------------------- | ------------------------------ | ---------------------------------------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600000` | Delt baseline for upstream-hentning, skjulte Undici-timeouts, TLS-fingeraftryksanmodninger og API-broanmodninger/proxy-timeouts | -| `STREAM_IDLE_TIMEOUT_MS` | arver `REQUEST_TIMEOUT_MS` | Maksimalt mellemrum mellem streamingstykker, før OmniRoute afbryder SSE-strømmen | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -Bagudkompatibilitet er bevaret: eksisterende `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` og andre timeoutvarianter pr. lag fungerer stadig og tilsidesætter den delte basislinje. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Avancerede tilsidesættelser er tilgængelige, hvis du har brug for bedre kontrol:| Variabel | Standard | Formål | -| ------------------------------------------ | ------------------------------------------ | ---------------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | arver `REQUEST_TIMEOUT_MS` | Total upstream-anmodningstimeout brugt af hovedhentningsafbrydelsessignalet | -| `FETCH_HEADERS_TIMEOUT_MS` | arver `FETCH_TIMEOUT_MS` | Undici tidsgrænse for modtagelse af opstrøms svaroverskrifter | -| `FETCH_BODY_TIMEOUT_MS` | arver `FETCH_TIMEOUT_MS` | Undici tidsgrænse mellem opstrøms kropsstykker (`0` deaktiverer det) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30.000` | Undici TCP forbindelse timeout | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | -| `TLS_CLIENT_TIMEOUT_MS` | arver `FETCH_TIMEOUT_MS` | Timeout for TLS-fingeraftryksanmodninger foretaget via `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | arver `REQUEST_TIMEOUT_MS` eller `30000` | Timeout for `/v1` proxy-videresendelse fra API-port til dashboard-port | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Timeout for indgående anmodning på API-broserveren | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60.000` | Timeout for indgående header på API-broserveren | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout på API-broserveren | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inaktivitet timeout på API-broserveren (`0` deaktiverer den) | +Advanced overrides are available if you need finer control: -Hvis du kører OmniRoute bag Nginx, Caddy, Cloudflare eller en anden omvendt proxy, skal du sørge for, at proxyen -timeouts er også højere end dine OmniRoute stream/hente timeouts.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Åbn Dashboard → `Providers` og tilslut mindst én udbyder (OAuth- eller API-nøgle). -2. Åbn Dashboard → `Endpoints` og opret en API-nøgle. -3. (Valgfrit) Åbn Dashboard → `Combos` og indstil din reservekæde.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Fungerer med Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode og OpenAI-kompatible SDK'er.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (til værktøjsdrevne operationer):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -Tilslut derefter din MCP-klient over 'stdio' og test værktøjer som: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (for agent-til-agent arbejdsgange):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -759,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Denne suite validerer rigtige MCP- og A2A-klientstrømme mod en kørende app.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -767,13 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` - -Ugyldig Linux (`xbps-src`-skabelon) +
+Void Linux (`xbps-src` template) -For Void Linux-brugere kan du bygge en indbygget pakke ved hjælp af `xbps-src`. Gem denne blok som `srcpkgs/omniroute/template`:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -785,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -793,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -869,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -880,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute er tilgængelig som et offentligt Docker-billede på [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Hurtigt løb:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -890,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**Med miljøfil:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Brug af Docker Compose:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Dashboard-understøttelse til Docker-implementeringer inkluderer nu et enkelt-klik**Cloudflare Quick Tunnel**på `Dashboard → Endpoints`. Den første aktivering downloader kun `cloudflared`, når det er nødvendigt, starter en midlertidig tunnel til dit nuværende `/v1`-slutpunkt og viser den genererede `https://*.trycloudflare.com/v1`-URL direkte under din normale offentlige URL. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Bemærkninger: +Notes: -- Hurtige tunnel-URL'er er midlertidige og ændres efter hver genstart. -- Hurtige tunneler gendannes ikke automatisk efter en OmniRoute- eller containergenstart. Genaktiver dem fra dashboardet, når det er nødvendigt. -- Administreret installation understøtter i øjeblikket Linux, macOS og Windows på `x64` / `arm64`. -- Managed Quick Tunnels er som standard HTTP/2-transport for at undgå støjende QUIC UDP-bufferadvarsler i begrænsede containermiljøer. Indstil `CLOUDFLARED_PROTOCOL=quic` eller `auto`, hvis du ønsker en anden transport. -- Docker-billeder samler systemets CA-rødder og sender dem til administreret `cloudflared`, hvilket undgår TLS-tillidsfejl, når tunnelen starter inde i containeren. -- SQLite kører i WAL-tilstand. `docker stop` skal have lov til at afslutte, så OmniRoute kan kontrollere de seneste ændringer tilbage i `storage.sqlite`. -- De medfølgende Compose-filer sætter allerede en 40'er-stop-periode. Hvis du kører billedet direkte, skal du beholde `--stop-timeout 40` (eller lignende), så manuelle stop ikke afbryder nedlukningsoprydning. -- Indstil `CLOUDFLARED_BIN=/absolute/sti/to/cloudflared`, hvis du ønsker, at OmniRoute skal bruge en eksisterende binær i stedet for at downloade en. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Brug af Docker Compose med Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute kan eksponeres sikkert ved hjælp af Caddys automatiske SSL-klargøring. Sørg for, at dit domænes DNS A-record peger på din servers IP.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` +| Image | Tag | Size | Description | +| ------------------------ | -------- | ------ | --------------------- | +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | -| Billede | Tag | Størrelse | Beskrivelse | -| -------------------------- | -------- | ------ | ---------------------- | -| `diegosouzapw/omniroute` | `nyeste` | ~250MB | Seneste stabile udgivelse | -| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Nuværende version |--- +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**NYT!**OmniRoute er nu tilgængelig som en**native desktop-applikation**til Windows, macOS og Linux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Kør OmniRoute som en selvstændig desktop-app - ingen terminal, ingen browser, intet internet påkrævet for lokale modeller. Den elektronbaserede app inkluderer: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Native Window**— Dedikeret appvindue med systembakkeintegration -- 🔄**Auto-Start**— Start OmniRoute ved systemlogin -- 🔔**Native notifikationer**— Få advarsler om kvoteopbrugt eller udbyderproblemer -- ⚡**One-Click Install**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Offline-tilstand**— Fungerer fuldt ud offline med medfølgende server### Kom hurtigt i gang +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Kom hurtigt i gang ```bash # Development mode @@ -979,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -Når den er minimeret, lever OmniRoute i din procesbakke med hurtige handlinger: +When minimized, OmniRoute lives in your system tray with quick actions: -- Åbn instrumentbrættet -- Skift serverport -- Afslut programmet +- Open dashboard +- Change server port +- Quit application -📖 Fuld dokumentation: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Tier | Udbyder | Omkostninger | Kvote nulstilling | Bedst til | -| ----------------- | --------------------------- | ------------------------------- | ------------------ | ------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -| **💳 ABONNEMENT** | Claude Code (Pro) | 20 USD/md. | 5 timer + ugentlig | Allerede abonneret | -| | Codex (Plus/Pro) | $20-200/md. | 5 timer + ugentlig | OpenAI-brugere | -| | Gemini CLI | **GRATIS** | 180K/md + 1K/dag | Alle sammen! | -| | GitHub Copilot | $10-19/md. | Månedlig | GitHub-brugere | -| **🔑 API NØGLE** | NVIDIA NIM | **GRATIS**(dev for evigt) | ~40 RPM | 70+ åbne modeller | -| | Cerebras | **GRATIS**(1M tok/dag) | 60K TPM / 30 RPM | Verdens hurtigste | -| | Groq | **GRATIS**(30 RPM) | 14,4K RPD | Ultrahurtig Lama/Gemma | -| | DeepSeek V3.2 | 0,27 USD/1,10 USD pr. 1 mio. | Ingen | Bedste pris/kvalitet ræsonnement | -| | xAI Grok-4 Hurtig | **$0,20/$0,50 pr. 1M**🆕 | Ingen | Hurtigste + værktøjsopkald, ultralav | -| | xAI Grok-4 (standard) | 0,20 USD/1,50 USD pr. 1 mio. 🆕 | Ingen | Fornuft flagskib fra xAI | -| | Mistral | Gratis prøveperiode + betalt | Sats begrænset | Europæisk AI | -| | OpenRouter | Betal pr. brug | Ingen | 100+ modeller aggr. | -| **💰 BILLIG** | GLM-5 (via Z.AI) 🆕 | 0,5 USD/1 mio. | Dagligt 10:00 | 128K output, nyeste flagskib | -| | GLM-4.7 | 0,6 USD/1 mio. | Dagligt 10:00 | Budget backup | -| | MiniMax M2.5 🆕 | $0,3/1 mio. input | 5-timers rullende | Begrundelse + agentopgaver | -| | MiniMax M2.1 | $0,2/1 mio. | 5-timers rullende | Billigste mulighed | -| | Kimi K2.5 (Moonshot API) 🆕 | Betal pr. brug | Ingen | Direkte Moonshot API-adgang | -| | Kimi K2 | 9 USD/md. lejlighed | 10M tokens/md. | Forudsigelige omkostninger | -| **🆓 GRATIS** | Qoder | **$0** | Ubegrænset | 5 modeller ubegrænset | -| | Qwen | **$0** | Ubegrænset | 4 modeller ubegrænset | -| | Kiro | **$0** | Ubegrænset | Claude Sonnet/Haiku (AWS Builder) | -| | LongCat Flash-Lite 🆕 | **$0**(50 mio. tok/dag 🔥) | 1 RPS | Største gratis kvote på jorden | -| | Bestøvninger AI 🆕 | **$0**(ingen nøgle nødvendig) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | -| | Cloudflare Workers AI 🆕 | **$0**(10.000 neuroner/dag) | ~150 hhv/dag | 50+ modeller, global kant | -| | Scaleway AI 🆕 | **$0**(1 mio. tokens i alt) | Sats begrænset | EU/GDPR, Qwen3 235B, Lama 70B | > 🆕**Nye modeller tilføjet (mars 2026):**Grok-4 Fast-familie til $0,20/$0,50/M (benchmarked ved 1143ms — 30 % hurtigere end Gemini 2.5 Flash), GLM-5 via Z.AI med 128K output, MiniMax M2.5-begrundelse, KimSeidek pr. direkte API. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 $0 Combo Stack — Den komplette gratis opsætning:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Nul omkostninger. Stopper aldrig med at kode.**Konfigurer dette som én OmniRoute-kombination, og alle fallbacks sker automatisk - ingen manuel skift nogensinde.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Alle modeller nedenfor er**100 % gratis uden kreditkort påkrævet**. OmniRoute dirigerer automatisk mellem dem, når én kvote løber ud - kombiner dem alle for en ubrydelig kombination af $0.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Model | Præfiks | Grænse | Satsgrænse | -| ------------------ | ------ | ------------- | ---------------------- | -| `claude-sonnet-4.5` | `kr/` |**Ubegrænset**| Ingen rapporteret dagligt loft | -| `claude-haiku-4.5` | `kr/` |**Ubegrænset**| Ingen rapporteret dagligt loft | -| `claude-opus-4.6` | `kr/` |**Ubegrænset**| Seneste Opus via Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) -| Model | Præfiks | Grænse | Satsgrænse | +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | --------------------- | +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | + +### 🟢 QODER MODELS (Free PAT via qodercli) + +| Model | Prefix | Limit | Rate Limit | | ------------------ | ------ | ------------- | --------------- | -| `kimi-k2-tænkning` | `hvis/` |**Ubegrænset**| Ingen rapporteret loft | -| `qwen3-coder-plus` | `hvis/` |**Ubegrænset**| Ingen rapporteret loft | -| `deepseek-r1` | `hvis/` |**Ubegrænset**| Ingen rapporteret loft | -| `minimax-m2.1` | `hvis/` |**Ubegrænset**| Ingen rapporteret loft | -| `kimi-k2` | `hvis/` |**Ubegrænset**| Ingen rapporteret loft | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -> Anbefalet forbindelsesmetode:**Personal Access Token + `qodercli`**. Browser OAuth er -> eksperimentel og deaktiveret som standard, medmindre `QODER_OAUTH_*` miljøvariabler er konfigureret.### 🟡 QWEN MODELS (Device Code Auth) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| Model | Præfiks | Grænse | Satsgrænse | -| ------------------ | ------ | ------------- | ------------------ | -| `qwen3-coder-plus` | `qw/` |**Ubegrænset**| Ingen rapporteret loft | -| `qwen3-coder-flash` | `qw/` |**Ubegrænset**| Ingen rapporteret loft | -| `qwen3-coder-next` | `qw/` |**Ubegrænset**| Ingen rapporteret loft | -| `vision-model` | `qw/` |**Ubegrænset**| Multimodal (billeder) |### 🟣 GEMINI CLI (Google OAuth) +### 🟡 QWEN MODELS (Device Code Auth) -| Model | Præfiks | Grænse | Satsgrænse | -| -------------------------- | ------ | -------------------------- | ------------- | -| `gemini-3-flash-preview` | `gc/` |**180K tok/måned**+ 1K/dag | Månedlig nulstilling | -| `gemini-2.5-pro` | `gc/` | 180K/måned (delt pool) | Høj kvalitet |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | ------------------- | +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Tier | Daglig grænse | Satsgrænse | Noter | -| ---------- | ------------ | ----------- | -------------------------------------------------------------- | -| Gratis (Dev) | Ingen token cap |**~40 RPM**| 70+ modeller; overgang til rene satsgrænser medio 2025 | +### 🟣 GEMINI CLI (Google OAuth) -Populære gratis modeller: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`/, `deepseek-seek`/`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -| Tier | Daglig grænse | Satsgrænse | Noter | -| ---- | ------------------ | ---------------- | -------------------------------------------------- | -| Gratis |**1 mio. tokens/dag**| 60K TPM / 30 RPM | Verdens hurtigste LLM-slutning; nulstilles dagligt | +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) -Tilgængelig gratis: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-destill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +| Tier | Daily Limit | Rate Limit | Notes | +| ---------- | ------------ | ----------- | ------------------------------------------------------ | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -| Tier | Daglig grænse | Satsgrænse | Noter | -| ---- | ------------- | ---------------- | ------------------------------------------ | -| Gratis |**14,4K RPD**| 30 RPM pr. model | Intet kreditkort; 429 på grænse, ikke opkrævet | +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -Tilgængelig gratis: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -| Model | Præfiks | Daglig gratis kvote | Noter | -| ------------------------------ | ------ | ------------------ | ---------------------------- | -| `LongCat-Flash-Lite` | `lc/` |**50 mio. tokens**💥 | Største gratis kvote nogensinde | -| `LongCat-Flash-Chat` | `lc/` | 500.000 tokens | Multi-turn chat | -| `LongCat-Flash-Thinking` | `lc/` | 500.000 tokens | Begrundelse / CoT | -| `LongCat-Flash-Thinking-2601` | `lc/` | 500.000 tokens | Jan 2026 version | -| `LongCat-Flash-Omni-2603` | `lc/` | 500.000 tokens | Multimodal | +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -> 100 % gratis, mens du er i offentlig beta. Tilmeld dig på [longcat.chat](https://longcat.chat) med e-mail eller telefon. Nulstiller dagligt 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -| Model | Præfiks | Satsgrænse | Udbyder bag | +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ------------- | ---------------- | ----------------------------------------- | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | + +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` + +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 + +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | + +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. + +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| `openai` | `pol/` | 1 req/15s | GPT-5 | -| `claude` | `pol/` | 1 req/15s | Antropiske Claude | -| `gemini` | `pol/` | 1 req/15s | Google Gemini | -| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | -| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | -| `mistral` | `pol/` | 1 req/15s | Mistral AI | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**Nul friktion:**Ingen tilmelding, ingen API-nøgle. Tilføj bestøvningsudbyderen med et tomt nøglefelt, og det virker med det samme.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| Tier | Daglige neuroner | Tilsvarende brug | Noter | -| ---- | ------------- | ----------------------------------------------- | ---------------------------- | -| Gratis |**10.000**| ~150 LLM resp. / 500s lyd / 15K indlejringer | Global kant, 50+ modeller | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 -Populære gratis modeller: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (gratis lyd!), `@cf/qwen/qwen2.5-coder-15b-instruct` +| Tier | Daily Neurons | Equivalent Usage | Notes | +| ---- | ------------- | --------------------------------------- | ----------------------- | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -> Kræver API-token + konto-id fra [dash.cloudflare.com](https://dash.cloudflare.com). Gem konto-id i udbyderindstillinger.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -| Tier | Gratis kvote | Beliggenhed | Noter | +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. + +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | | ---- | ------------- | ------------ | ----------------------------------- | -| Gratis |**1 mio. tokens**| 🇫🇷 Paris, EU | Intet kreditkort nødvendigt inden for grænserne | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | -Tilgængelig gratis: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` -> EU/GDPR-kompatibel. Hent API-nøgle på [console.scaleway.com](https://console.scaleway.com). +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). ->**💡 Den ultimative gratis stak (11 udbydere, $0 for evigt):** +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -> Qoder (hvis/) → kimi-k2-tænkning, qwen3-coder-plus, deepseek-r1 UNLIMITED -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 mio. tokens/dag 🔥 -> Bestøvninger (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — ingen nøgle nødvendig -> Qwen (qw/) → qwen3-koder modeller UBEGRÆNSET -> Gemini (gemini/) → Gemini 2.5 Flash — 1.500 req/dag gratis -> Cloudflare AI (jf/) → 50+ modeller — 10K neuroner/dag -> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M gratis tokens (EU) -> Groq (groq/) → Lama/Gemma — 14,4K req/dag ultrahurtig -> NVIDIA NIM (nvidia/) → 70+ åbne modeller — 40 RPM for evigt -> Cerebras (cerebras/) → Lama/Qwen verdenshurtigste — 1M tok/dag -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Transskriber enhver lyd/video for**$0**— Deepgram-emner med $200 gratis, AssemblyAI $50 fallback, Groq Whisper som ubegrænset nødbackup. +## 🎙️ Free Transcription Combo -| Udbyder | Gratis kreditter | Bedste model | Satsgrænse | -| ------------------ | ---------------------- | -------------------------------------------- | ---------------------------- | -|**Deepgram**|**$200 gratis**(tilmelding) | `nova-3` — bedste nøjagtighed, 30+ sprog | Ingen RPM-grænse på gratis kreditter | -| 🔵**AssemblyAI**|**$50 gratis**(tilmelding) | `universal-3-pro` — kapitler, følelser, PII | Ingen RPM-grænse på gratis kreditter | -| 🔴**Groq**|**Gratis for evigt**| `whisper-large-v3` — OpenAI Whisper | 30 RPM (hastighedsbegrænset) | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. -**Foreslået kombination i `/dashboard/combos`:**``` +| Provider | Free Credits | Best Model | Rate Limit | +| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | + +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Derefter i `/dashboard/media` → fanen**Transskription**: upload en lyd- eller videofil → vælg dit kombinationsslutpunkt → få transskription i understøttede formater.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 er bygget som en operationel platform, ikke kun en relæ-proxy.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Funktion | Hvad det gør | -| ----------------------------------------- | ----------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Grok-4 Fast Family** | xAI-modeller til $0,20/$0,50/M — benchmarked 1143ms (30 % hurtigere end Gemini 2,5 Flash) | -| 🧠**GLM-5 via Z.AI** | 128K output kontekst, $0,5/1M — nyeste flagskib fra GLM-familien | -| 🔮**MiniMax M2.5** | Begrundelse + agentopgaver til $0,30/1M — betydelig opgradering fra M2.1 | -| 🎯**værktøj Calling Flag per model** | "ToolCalling" pr. model: sand/falsk i registreringsdatabasen — AutoCombo springer ikke-værktøjskompatible modeller over | -| 🌍**Flersproget hensigtsdetektion** | PT/ZH/ES/AR nøgleord i AutoCombo scoring — bedre modelvalg for ikke-engelsk indhold | -| 📊**Benchmark-drevne fallbacks** | Ægte p95-forsinkelse fra live-anmodninger feeds combo scoring — AutoCombo lærer af faktiske data | -| 🔁**Anmod om deduplikation** | Indholdshash-baseret dedup-vindue — multi-agent sikker, forhindrer duplikerede debiteringer | -| 🔌**Strategi, der kan tilsluttes router** | Udvidelig `RouterStrategy`-grænseflade — tilføj brugerdefineret routinglogik som plugins | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Funktion | Hvad det gør | -| ----------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Model Legeplads** | Dashboard-side for at teste enhver model direkte — udbyder/model/slutpunktsvælgere, Monaco Editor, streaming, afbrydelse, timing | -| 🔏**CLI Fingerprint Matching** | Bestilling af header/body pr. udbyder for at matche native CLI-signaturer — skift pr. udbyder i Indstillinger > Sikkerhed.**Din proxy-IP er bevaret** | -| 🤝**ACP Support (Agent Client Protocol)** | CLI-agentopdagelse (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 mere), procesopstart, `/api/acp/agents` slutpunkt | -| 🤖**ACP Agents Dashboard** | Fejlfinding › Agenter-side — gitter med 14 agenter med installationsstatus, version, brugerdefineret agentformular til ethvert CLI-værktøj.**OpenCode**-brugere får en "Download opencode.json"-knap, der automatisk genererer en klar-til-brug-konfiguration med alle tilgængelige modeller. | -| 🔧**Brugerdefineret model `apiFormat` Routing** | Brugerdefinerede modeller med `apiFormat: "responses"` rutes nu korrekt til Responses API-oversætteren | -| 🏢**Codex Workspace Isolation** | Flere Codex-arbejdsområder pr. e-mail — OAuth adskiller forbindelser korrekt efter arbejdsområde-id | -| 🔄**Automatisk opdatering af elektroner** | Desktop-app søger efter opdateringer + automatisk installation ved genstart | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Funktion | Hvad det gør | -| --------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**MCP-server (25 værktøjer)** | IDE/agent værktøjer via 3 transporter: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 kerner + 3 hukommelse + 4 færdighedsværktøjer | -| 🤝**A2A-server (JSON-RPC + SSE)** | Agent-til-agent opgaveudførelse med synkronisering og streaming flows | -| 🧭**Konsoliderede slutpunkter-side** | Administrationsside med faner med Endpoint Proxy, MCP, A2A og API Endpoints faner | -| 🎚️**Tjenesteaktiver/deaktiver skifter** | ON/OFF-kontakter til MCP og A2A med fastholdelse af indstillinger (standard: OFF) | -| 🛰️**MCP Runtime Heartbeat** | Reel processtatus (pid, oppetid, hjerteslagsalder, transport, omfangstilstand) | -| 📋**MCP Audit Trail** | Filtrerbare revisionslogfiler med succes/fejl og nøgletilskrivning | -| 🔐**MCP Scope Enforcement** | 10 granulære omfangstilladelser til kontrolleret værktøjsadgang | -| 📡**A2A Task Lifecycle Management** | Liste/filtrere opgaver, inspicere hændelser/artefakter, annullere kørende opgaver | -| 📋**Agent Card Discovery** | `/.well-known/agent.json` til klient auto-discovery | -| 🧪**Protokol E2E testsele** | Ægte MCP SDK + A2A klient flows i `test:protocols:e2e` | -| ⚙️**Driftskontrol** | Switch combo, påfør elasticitetsprofiler, nulstil afbrydere fra én kontrolflade | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Funktion | Hvad det gør | -| ---------------------------------- | ------------------------------------------------------------------------------- | ----------------------- | -| 🎯**Smart 4-lags fallback** | Auto-rute: Abonnement → API-nøgle → Billig → Gratis | -| 📊**Kvotesporing i realtid** | Live token count + nulstil nedtælling pr. udbyder | -| 🔄**Formatoversættelse** | OpenAI ↔ Claude ↔ Gemini ↔ Svar med skemasikre konverteringer | -| 👥**Multi-Account Support** | Flere konti pr. udbyder med intelligent valg | -| 🔄**Automatisk token-opdatering** | OAuth-tokens opdateres automatisk med genforsøg | -| 🎨**Tilpassede kombinationer** | 9 balanceringsstrategier + fallback kædekontrol | -| 🌐**Wildcard-router** | `udbyder/*` dynamisk routing | -| 🧠**Tænker på budgetkontrol** | Grænser for gennemstrømning, automatisk, brugerdefineret og adaptiv ræsonnement | -| 🔀**Modelaliaser** | Indbygget + brugerdefineret model aliasing og migration sikkerhed | -| ⚡**Baggrundsforringelse** | Send baggrundsopgaver med lav prioritet til billigere modeller | -| 🧪**Task-Aware Smart Routing** | Auto-vælg model efter indholdstype (kodning/vision/analyse/opsummering) | -| 🔄**A2A Agent Workflows** | Deterministisk FSM-orkestrator til stateful multi-step agent henrettelser | -| 🔀**Adaptiv Routing** | Dynamisk strategitilsidesættelse baseret på tokenvolumen og promptkompleksitet | -| 🎲**Udbyderdiversitet** | Shannon entropi-scoring balancerer auto-combo-trafikfordeling | -| 💬**System Prompt Injection** | Globale adfærdskontroller anvendes konsekvent | -| 📄**Responses API-kompatibilitet** | Fuld `/v1/responses`-understøttelse af Codex og avancerede agent-arbejdsgange | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Funktion | Hvad det gør | -| ----------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**Billedgenerering** | `/v1/images/generations` med sky og lokale backends | -| 📐**Indlejringer** | `/v1/embeddings` til søgning og RAG-rørledninger | -| 🎤**Lydtransskription** | `/v1/audio/transcriptions` — 7 udbydere (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-sprogdetektion, MP4/MP3/WAV-understøttelse | -| 🔊**Tekst-til-tale** | `/v1/audio/speech` — 10 udbydere (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) med korrekte fejlmeddelelser | -| 🎬**Videogenerering** | `/v1/videos/generations` (ComfyUI + SD WebUI-arbejdsgange) | -| 🎵**Music Generation** | `/v1/music/generations` (ComfyUI-arbejdsgange) | -| 🛡️**Moderationer** | `/v1/moderations` sikkerhedstjek | -| 🔀**Omrangering** | `/v1/rerank` for relevansscoring | -| 🔍**Websøgning**🆕 | `/v1/search` — 5 udbydere (Serper, Brave, Perplexity, Exa, Tavily), 6.500+ gratis/måned, auto-failover, cache | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Funktion | Hvad det gør | -| ----------------------------------- | ------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**Maksimalafbrydere** | Pr. model tur/restitution med tærskelkontrol | -| 🎯**Endpoint-Aware-modeller** | Brugerdefinerede modeller erklærer understøttede slutpunkter + API-format | -| 🛡️**Anti-tordenbesætning** | Mutex + semaforbeskyttelse ved genforsøg/rate hændelser | -| 🧠**Semantisk + signaturcache** | Reduktion af omkostninger/latens med to cachelag | -| ⚡**Anmod om idempotens** | Dobbelt beskyttelsesvindue | -| 🔒**TLS Fingerprint Spoofing** | Browserlignende TLS-fingeraftryk —**reducerer botgenkendelse og kontoflaggning** | -| 🔏**CLI Fingerprint Matching** | Matcher native CLI-anmodningssignaturer —**reducerer forbudsrisiko, mens proxy-IP bevares** | -| 🌐**IP-filtrering** | Tilladelsesliste/blokeringslistekontrol for udsatte implementeringer | -| 📊**Redigerbare satsgrænser** | Konfigurerbare grænser på globalt niveau/udbyderniveau med persistens | -| 📉**Graceful Nedbrydning** | Muligheder med flere lag, der beskytter kerne-gateway-operationer | -| 📜**Config Audit Trail** | Diff-baseret ændringssporing forhindrer driftsafdrift med simple rollbacks | -| ⏳**Provider Health Sync** | Proaktiv overvågning af tokens udløb, der udløser advarsler før godkendelsesfejl | -| 🚪**Auto-deaktiver forbudte konti** | Driftsafbryder forsegling permanent blokerede token-konti automatisk | -| 🔑**API Key Management + Scoping** | Sikker nøgleudstedelse/rotation og model-/leverandørkontrol | -| 👁️**Scoped API Key Reveal**🆕 | Opt-in gendannelse af API-nøgler via `ALLOW_API_KEY_REVEAL` | -| 🛡️**Beskyttet `/modeller`** | Valgfri godkendelse og udbyderskjul til modelkatalog | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Funktion | Hvad det gør | -| -------------------------------------- | ---------------------------------------------------------------- | ---------------------------- | -| 📝**Forespørgsel + Proxylogning** | Fuld anmodning/svar og proxy-logning | -| 📉**Streamede detaljerede logfiler**🆕 | Rekonstruerer SSE-nyttelaststrømme rent ind i brugergrænsefladen | -| 📋**Unified Logs Dashboard** | Anmodning, proxy, revision og konsolvisning på én side | -| 🔍**Anmod om telemetri** | p50/p95/p99 latens og anmodningssporing | -| 🏥**Sundhedskontrolpanel** | Oppetid, breaker-tilstande, lockouts, cache-statistik | -| 💰**Omkostningssporing** | Budgetkontrol og prisfastsættelse pr. model | -| 📈**Analytiske visualiseringer** | Model-/udbyderbrugsindsigt og trendvisninger | -| 🧪**Evalueringsramme** | Gyldne sæt-test med konfigurerbare matchstrategier | -| 📡**Live Diagnostics**🆕 | Semantisk cache-bypass for nøjagtig combo live-test | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Funktion | Hvad det gør | -| ------------------------------------- | ----------------------------------------------------------------- | --------------------- | -| 🌐**Deploy hvor som helst** | Localhost, VPS, Docker, Cloud-miljøer | -| 🚇**Cloudflare Tunnel**🆕 | Hurtig tunnel-integration med et enkelt klik fra dashboardet | -| 🔑**API-nøglemodelfiltrering** | Native /v1/models-svar filtreret via tildelte bærerkontekstroller | -| ⚡**Smart Cache Bypass** | Konfigurerbar TTL-heuristik og tvungen genhentningskontroller | -| 🔄**Sikkerhedskopiering/gendannelse** | Eksport/import og gendannelsesstrømme | -| 🧙**Onboarding Wizard** | Første kørsel guidet opsætning | -| 🔧**CLI Tools Dashboard** | Et-klik opsætning til populære kodningsværktøjer | -| 🎮**Model Legeplads** | Test enhver udbyder/model/slutpunkt fra dashboardet | -| 🔏**CLI Fingerprint Toggle** | Fingeraftryksmatchning pr. udbyder i Indstillinger > Sikkerhed | -| 🌐**i18n (30 sprog)** | Fuldt dashboard + understøttelse af docs-sprog med RTL-dækning | -| 🧹**Ryd alle modeller** | Rydning af modelliste med ét klik i udbyderoplysninger | -| 👁️**Sidebjælkekontrol**🆕 | Skjul komponenter og integrationer fra Udseendeindstillinger | -| 📋**Udgaveskabeloner** | Standardiserede GitHub-skabeloner til fejl og funktioner | -| 📂**Tilpasset datakatalog** | `DATA_DIR` tilsidesættelse for lagerplacering | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1292,103 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -Når kvote, sats eller sundhed svigter, flytter OmniRoute automatisk til den næste kandidat uden manuel skift.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A kan findes i brugergrænsefladen og dokumenter (ikke skjult) -- Protokolstatus API'er afslører live operationelle data (`/api/mcp/*`, `/api/a2a/*`) -- Dashboards inkluderer handlinger for dag-2 operationer (kombinationsskift, nulstilling af breaker, annullering af opgave)#### Translator + validation workflow +#### Protocol management that is visible and operable -Oversætterområdet omfatter: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Legeplads**: anmod om transformationstjek -**Chattester**: fuld anmodning/svar tur/retur -**Testbænk**: flere sager på én gang -**Live Monitor**: trafikvisning i realtid +#### Translator + validation workflow -Plus protokolvalidering med rigtige klienter via `npm run test:protocols:e2e`. +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Værktøjsreference, IDE-konfigurationer og klienteksempler +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A Server README](src/lib/a2a/README.md)**— Færdigheder, JSON-RPC-metoder, streaming og opgavelivscyklus## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute inkluderer en indbygget evalueringsramme til at teste LLM-svarkvaliteten mod et gyldent sæt. Få adgang til det via**Analytics → Evals**i dashboardet.### Built-in Golden Set +## 🧪 Evaluations (Evals) -Det forudindlæste "OmniRoute Golden Set" indeholder testcases til: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Hilsen, matematik, geografi, kodegenerering -- JSON format compliance, oversættelse, markdown generation -- Sikkerhedsafvisning (skadeligt indhold), optælling, boolsk logik### Evaluation Strategies +### Built-in Golden Set -| Strategi | Beskrivelse | Eksempel | -| ----------------- | ----------------------------------------------------------------------- | -------------------------------- | --- | -| 'præcis' | Output skal matche nøjagtigt | `"4"` | -| `indeholder` | Output skal indeholde understreng (uafhængig af store og små bogstaver) | `"Paris"` | -| "regex" | Output skal matche regex-mønster | `"1.*2.*3"` | -| `brugerdefineret` | Brugerdefineret JS-funktion returnerer sand/falsk | `(output) => output.længde > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) - -🧩 MCP-opsætning (modelkontekstprotokol) +
+🧩 MCP Setup (Model Context Protocol) -Start MCP-transport i stdio-tilstand:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Anbefalet valideringsflow: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Tilslut din MCP-klient via stdio. -2. Kør `omniroute_get_health`. -3. Kør `omniroute_list_combos`. -4. Åbn `/dashboard/mcp` for at bekræfte hjerteslag, aktivitet og audit. - -Nyttige API'er til automatisering: +Useful APIs for automation: - `GET /api/mcp/status` - `GET /api/mcp/tools` - `GET /api/mcp/audit` -- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats` - -🤝 A2A-opsætning (Agent2Agent) + -Opdag agenten:```bash +
+🤝 A2A Setup (Agent2Agent) + +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Send en opgave:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` - -Administrer livscyklus: +Manage lifecycle: - `GET /api/a2a/status` -- `GET /api/a2a/opgaver` +- `GET /api/a2a/tasks` - `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -Operationel UI: +Operational UI: -- `/dashboard/a2a` til observerbarhed for opgave/tilstand/strøm og røghandlinger
+- `/dashboard/a2a` for task/state/stream observability and smoke actions - -🧪 End-to-end protokolvalidering + -Valider begge protokoller med rigtige klienter:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Dette verificerer: +This verifies: -- MCP SDK-klient forbinde/liste/opkald -- A2A opdagelse/send/stream/hent/annuller -- Krydstjek data i MCP-audit og A2A opgavestyring API'er
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs - -💳 Abonnementsudbydere### Claude Code (Pro/Max) + + +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1401,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Prof tip:**Brug Opus til komplekse opgaver, Sonnet for hurtighed. OmniRoute sporer kvote pr. model!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1415,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Hver Codex-konto har nu politikskift i `Dashboard -> Udbydere`: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (ON/OFF): håndhæv politikken for 5-timers vinduestærskel. -- `Ugentligt` (TIL/FRA): håndhæv politikken for tærskelværdi for ugentlige vinduer. -- Tærskeladfærd: Når et aktiveret vindue når >=90 % brug, springes den konto over. -- Rotationsadfærd: OmniRoute ruter automatisk til den næste kvalificerede Codex-konto. -- Nulstil adfærd: Når udbyderens 'resetAt'-tid går, bliver kontoen automatisk kvalificeret igen. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Scenarier: +Scenarios: -- `5h ON` + ` Weekly ON`: Konto springes over, når et af vinduerne når tærsklen. -- `5h OFF` + ` Weekly ON`: kun ugentlig brug kan blokere kontoen. -- `5h ON` + `Ugentlig OFF`: kun 5-timers brug kan blokere kontoen. -- `resetAt` bestået: Kontoen går automatisk i rotation igen (ingen manuel genaktivering).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1440,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Bedste værdi:**Kæmpe gratis niveau! Brug dette før betalte niveauer.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1455,71 +1662,91 @@ Models:
- -🔑 API-nøgleudbydere### NVIDIA NIM (FREE developer access — 70+ models) +
+🔑 API Key Providers -1. Tilmeld dig: [build.nvidia.com](https://build.nvidia.com) -2. Få gratis API-nøgle (1000 slutningskreditter inkluderet) -3. Dashboard → Tilføj udbyder → NVIDIA NIM: - - API-nøgle: `nvapi-din-nøgle` +### NVIDIA NIM (FREE developer access — 70+ models) -**Modeller:**"nvidia/llama-3.3-70b-instruct", "nvidia/mistral-7b-instruct" og 50+ flere +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Prof tip:**OpenAI-kompatibel API — fungerer problemfrit med OmniRoutes formatoversættelse!### DeepSeek +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -1. Tilmeld dig: [platform.deepseek.com](https://platform.deepseek.com) -2. Hent API-nøgle -3. Dashboard → Tilføj udbyder → DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -**Modeller:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +### DeepSeek -1. Tilmeld dig: [console.groq.com](https://console.groq.com) -2. Få API-nøgle (gratis niveau inkluderet) -3. Dashboard → Tilføj udbyder → Groq +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -**Modeller:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Prof tip:**Ultrahurtig inferens — bedst til realtidskodning!### OpenRouter (100+ Models) +### Groq (Free Tier Available!) -1. Tilmeld dig: [openrouter.ai](https://openrouter.ai) -2. Hent API-nøgle -3. Dashboard → Tilføj udbyder → OpenRouter +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -**Modeller:**Få adgang til mere end 100 modeller fra alle større udbydere via en enkelt API-nøgle. +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Dashboard-adfærd:**OpenRouter-modeller administreres fra**Tilgængelige modeller**. Manuel tilføjelse, import og automatisk synkronisering opdaterer alle den samme liste.
+**Pro Tip:** Ultra-fast inference — best for real-time coding! - -💰 Billige udbydere (backup)### GLM-4.7 (Daily reset, $0.6/1M) +### OpenRouter (100+ Models) -1. Tilmeld dig: [Zhipu AI](https://open.bigmodel.cn/) -2. Hent API-nøgle fra Coding Plan -3. Dashboard → Tilføj API-nøgle: - - Udbyder: `glm` - - API-nøgle: `din-nøgle` +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -**Brug:**`glm/glm-4.7` +**Models:** Access 100+ models from all major providers through a single API key. -**Pro-tip:**Coding Plan tilbyder 3× kvote til 1/7 pris! Nulstil dagligt 10:00.### MiniMax M2.1 (5h reset, $0.20/1M) +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -1. Tilmeld dig: [MiniMax](https://www.minimax.io/) -2. Hent API-nøgle -3. Dashboard → Tilføj API-nøgle + -**Brug:**`minimax/MiniMax-M2.1` +
+💰 Cheap Providers (Backup) -**Prof tip:**Billigste mulighed for lang sammenhæng (1M tokens)!### Kimi K2 ($9/month flat) +### GLM-4.7 (Daily reset, $0.6/1M) -1. Abonner: [Moonshot AI](https://platform.moonshot.ai/) -2. Hent API-nøgle -3. Dashboard → Tilføj API-nøgle +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**Brug:**`kimi/kimi-latest` +**Use:** `glm/glm-4.7` -**Prof tip:**Fast $9/måned for 10M tokens = $0,90/1M effektive omkostninger!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. - -🆓 GRATIS udbydere (nødbackup)### Qoder (5 FREE models via OAuth) +### MiniMax M2.1 (5h reset, $0.20/1M) + +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `minimax/MiniMax-M2.1` + +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1560,8 +1787,10 @@ Models:
- -🎨 Opret kombinationer### Example 1: Maximize Subscription → Cheap Backup +
+🎨 Create Combos + +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1589,8 +1818,10 @@ Cost: $0 forever!
- -🔧 CLI-integration### Cursor IDE +
+🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1601,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Brug siden**CLI Tools**i dashboardet til konfiguration med et enkelt klik, eller rediger `~/.claude/settings.json` manuelt.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1612,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Mulighed 1 — Dashboard (anbefalet):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Mulighed 2 — Manuel:**Rediger `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1629,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Bemærk:**OpenClaw fungerer kun med lokale OmniRoute. Brug `127.0.0.1` i stedet for `localhost` for at undgå problemer med IPv6-opløsning.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1643,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Trin 1:**Tilføj OmniRoute som en tilpasset udbyder:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Trin 2:**Opret/rediger `opencode.json` i dit projektrod:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1669,117 +1909,130 @@ opencode } } } -```` +``` -**Trin 3:**Vælg modellen i OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Tip:**Tilføj en hvilken som helst model, der er tilgængelig i dit OmniRoute `/v1/models` slutpunkt til sektionen `modeller`. Brug formatet `provider/model-id` fra dit OmniRoute-dashboard.
+ --- ## Fejlfinding - -Klik for at udvide fejlfindingsvejledningen +
+Click to expand troubleshooting guide -**"Sprogmodellen leverede ikke beskeder"** +**"Language model did not provide messages"** -- Udbyderkvote opbrugt → Tjek dashboardkvotesporing -- Løsning: Brug combo fallback eller skift til et billigere niveau +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**Satsbegrænsende** +**Rate limiting** -- Abonnementskontingent ude → Fallback til GLM/MiniMax -- Tilføj combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**OAuth-token er udløbet** +**OAuth token expired** -- Automatisk genopfrisket af OmniRoute -- Hvis problemerne fortsætter: Dashboard → Udbyder → Genopret forbindelse +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Høje omkostninger** +**High costs** -- Tjek brugsstatistik i Dashboard → Omkostninger -- Skift primær model til GLM/MiniMax -- Brug gratis niveau (Gemini CLI, Qoder) til ikke-kritiske opgaver +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Dashboard/API-porte er forkerte** +**Dashboard/API ports are wrong** -- `PORT` er den kanoniske basisport (og API-port som standard) -- `API_PORT` tilsidesætter kun OpenAI-kompatibel API-lytter -- `DASHBOARD_PORT` tilsidesætter kun dashboard/Next.js-lytter -- Indstil `NEXT_PUBLIC_BASE_URL` til dit dashboard/offentlige URL (til OAuth-tilbagekald) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Skysynkroniseringsfejl** +**Cloud sync errors** -- Bekræft, at `BASE_URL` peger på din kørende instans -- Bekræft `CLOUD_URL` peger på dit forventede cloud-slutpunkt -- Hold `NEXT_PUBLIC_*`-værdier på linje med værdier på serversiden +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Første login virker ikke** +**First login not working** -- Tjek `INITIAL_PASSWORD` i `.env` -- Hvis den ikke er indstillet, er reserveadgangskoden "123456". +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Ingen anmodningslogfiler** +**No request logs** -- Anmodningsartefakter skrives til `DATA_DIR/call_logs/` som én JSON-fil pr. anmodning -- Aktiver pipeline capture fra Dashboard → Logs → Request Logs, hvis du har brug for detaljerede per-stage payloads -- Indstil `APP_LOG_TO_FILE=true`, hvis du også vil have applikationskonsollogfiler i `logs/application/app.log` -- Juster `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` og `CALL_LOG_MAX_ENTRIES` efter behov +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Forbindelsestest viser "Ugyldig" for OpenAI-kompatible udbydere** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Mange udbydere afslører ikke et `/models` slutpunkt -- OmniRoute v1.0.6+ inkluderer fallback-validering via chatafslutninger -- Sørg for, at basis-URL'en indeholder `/v1`-suffiks### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server - + ->**⚠️ Vigtigt for brugere, der kører OmniRoute på en VPS, Docker eller enhver ekstern server**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -**Antigravity**og**Gemini CLI**-udbyderne bruger**Google OAuth 2.0**. Google kræver, at `redirect_uri` i OAuth-flowet nøjagtigt matcher en af ​​de forudregistrerede URI'er i appens Google Cloud Console. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -OAuth-legitimationsoplysningerne, der er bundtet i OmniRoute, er kun registreret**for 'localhost'**. Når du får adgang til OmniRoute på en ekstern server (f.eks. `https://omniroute.myserver.com`), afviser Google godkendelsen med:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Du skal oprette et**OAuth 2.0 Client ID**i Google Cloud Console med din servers URI.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Åbn Google Cloud Console** +#### Step-by-step -Gå til: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Opret et nyt OAuth 2.0-klient-id** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Klik på**"+ Opret legitimationsoplysninger"**→**"OAuth-klient-id"** -- Ansøgningstype:**"Webapplikation"** -- Navn: alt, hvad du kan lide (f.eks. `OmniRoute Remote`) +**2. Create a new OAuth 2.0 Client ID** -**3. Tilføj autoriserede omdirigerings-URI'er** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -I feltet**"Autoriserede omdirigerings-URI'er"**skal du tilføje:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Erstat `din-server.com` med din servers domæne eller IP (medtag porten, hvis det er nødvendigt, f.eks. `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. Gem og kopier legitimationsoplysningerne** +After creating, Google will show the **Client ID** and **Client Secret**. -Efter oprettelse vil Google vise**klient-id**og**klienthemmelighed**. +**5. Set environment variables** -**5. Indstil miljøvariabler** +In your `.env` (or Docker environment variables): -I dine `.env` (eller Docker-miljøvariabler):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1788,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Genstart OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Prøv at oprette forbindelse igen** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Dashboard → Udbydere → Antigravity (eller Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google vil nu omdirigere korrekt til `https://din-server.com/callback`.--- +--- #### Temporary workaround (without custom credentials) -Hvis du ikke vil konfigurere dine egne legitimationsoplysninger lige nu, kan du stadig bruge det**manuelle URL-flow**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute åbner Googles autorisations-URL -2. Efter godkendelse forsøger Google at omdirigere til `localhost` (som fejler på fjernserveren) -3.**Kopiér den fulde URL**fra din browsers adresselinje (også selvom siden ikke indlæses) -4. Indsæt denne URL i feltet vist i OmniRoute-forbindelsesmodal -5. Klik på**"Forbind"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Dette virker, fordi autorisationskoden i URL'en er gyldig, uanset om omdirigeringssiden er indlæst.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. - -🇧🇷 Versão em Português#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Os testedores**Antigravity**og**Gemini CLI**usam**Google OAuth 2.0**for autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja**exatamente**uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. +
+🇧🇷 Versão em Português -Som credenciais OAuth embutidas no OmniRoute estão cadastradas**apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google afviser en autenticação com:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Você precisa criar um**OAuth 2.0 Client ID**ingen Google Cloud Console med en URI, der udfører denne service.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Adgang til Google Cloud Console** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) **2. Crie um novo OAuth 2.0 Client ID** -- Klik på dem**"+ Opret legitimationsoplysninger"**→**"OAuth-klient-id"** -- Tipo de aplicativo:**"Webapplikation"** -- Navn: escolha qualquer nome (eks.: `OmniRoute Remote`) +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Adicione som autoriseret omdirigerings-URI** +**3. Adicione as Authorized Redirect URIs** -Ingen campo**"Autoriseret omdirigerings-URI'er"**, adicione:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Substitua `seu-servidor.com` pelo domínio eller IP do seu servidor (inklusive en porta se necessário, f.eks: `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. Salve e copy as credenciais** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Após criar, o Google mostrará o**Client ID**e o**Client Secret**. +**5. Configure as variáveis de ambiente** -**5. Konfigurer som variáveis de ambiente** +No seu `.env` (ou nas variáveis de ambiente do Docker): -No seu `.env` (ou nas variáveis de ambiente do Docker):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1867,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reinicie o OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute - -```` +``` **7. Tente conectar novamente** -Dashboard → Udbydere → Antigravity (ou Gemini CLI) → OAuth +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` og autenticação funcionará.--- +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. + +--- #### Workaround temporário (sem configurar credenciais próprias) -Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo**manual de URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. O OmniRoute abrirá en URL de autorização til Google -2. Após você autorizar, o Google tentará redirecionar for `localhost` (que falha no servidor remoto) -3.**Kopier en URL komplet**da barra de endereço do sin browser (mesmo que a página não carregue) +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) 4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute -5. Klik på**"Forbind"** +5. Clique em **"Connect"** -> Este workaround funciona porque or código de autorização na URL é válido independente do redirect ter carregado or não.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1905,64 +2171,73 @@ Se não quiser criar credenciais próprias agora, ainda é possível usar o flux ## 🛠️ Tech Stack - -Klik for at udvide tekniske stakdetaljer +
+Click to expand tech stack details --**Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ er**ikke understøttet**— "better-sqlite3" native binære filer er inkompatible) --**Sprog**: TypeScript 5.9 —**100 % TypeScript**på tværs af `src/` og `open-sse/` (nul `enhver` i kernemoduler siden v2.0) --**Framework**: Next.js 16 + React 19 + Tailwind CSS 4 --**Database**: LowDB (JSON) + SQLite (domænetilstand + proxylogfiler + MCP-revision + routingbeslutninger) --**Skemaer**: Zod (MCP-værktøj I/O-validering, API-kontrakter) --**Protokoller**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Streaming**: Server-sendte hændelser (SSE) --**Auth**: OAuth 2.0 (PKCE) + JWT + API-nøgler + MCP Scoped Authorization --**Test**: Node.js testløber + Vitest (900+ tests inklusive enhed, integration, E2E) --**CI/CD**: GitHub-handlinger (automatisk npm-udgivelse + Docker Hub ved udgivelse) --**Websted**: [omniroute.online](https://omniroute.online) --**Pakke**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Resiliens**: Circuit breaker, eksponentiel backoff, anti-tordenbesætning, TLS spoofing, auto-combo selvhelbredelse
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Dokumentation -| Dokument | Beskrivelse | -| ------------------------------------------------------ | ---------------------------------------------------------- | -| [Brugervejledning](docs/USER_GUIDE.md) | Udbydere, kombinationer, CLI-integration, implementering | -| [API-reference](docs/API_REFERENCE.md) | Alle endepunkter med eksempler | -| [MCP-server](open-sse/mcp-server/README.md) | 16 MCP-værktøjer, IDE-konfigurationer, Python/TS/Go-klienter | -| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protokol, færdigheder, streaming, opgavestyring | -| [Auto-Combo Engine](docs/auto-combo.md) | 6-faktor scoring, tilstandspakker, selvhelbredende | -| [Fejlfinding](docs/TROUBLESHOOTING.md) | Almindelige problemer og løsninger | -| [Arkitektur](docs/ARCHITECTURE.md) | Systemarkitektur og indre | -| [Bidrager](BIDRØRENDE.md) | Udviklingsopsætning og retningslinjer | -| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0-specifikation | -| [Sikkerhedspolitik](SECURITY.md) | Sårbarhedsrapportering og sikkerhedspraksis | -| [VM-implementering](docs/VM_DEPLOYMENT_GUIDE.md) | Komplet guide: VM + nginx + Cloudflare opsætning | -| [Feature Gallery](docs/FEATURES.md) | Visuel dashboard-rundvisning med skærmbilleder | -| [Udgivelsestjekliste](docs/RELEASE_CHECKLIST.md) | Pre-release valideringstrin |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute har**210+ funktioner planlagt**på tværs af flere udviklingsfaser. Her er nøgleområderne: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Kategori | Planlagte funktioner | Højdepunkter | -| ------------------------------ | ---------------- | ---------------------------------------------------------------------------------------------- | -| 🧠**Routing & intelligens**| 25+ | Routing med laveste latens, tag-baseret routing, kvote preflight, valg af P2C-konto | -| 🔒**Sikkerhed og overholdelse**| 20+ | SSRF-hærdning, tilsløring af legitimationsoplysninger, hastighedsgrænse pr. slutpunkt, styringsnøgleomfang | -| 📊**Observabilitet**| 15+ | OpenTelemetry-integration, kvoteovervågning i realtid, omkostningssporing pr. model | -| 🔄**Udbyderintegrationer**| 20+ | Dynamisk modelregistrering, udbydernedkøling, multi-konto Codex, Copilot-kvoteparsing | -| ⚡**Ydeevne**| 15+ | Dobbelt cachelag, promptcache, svarcache, streaming keepalive, batch API | -| 🌐**Økosystem**| 10+ | WebSocket API, config hot-reload, distribueret config butik, kommerciel tilstand |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**OpenCode-integration**— Native udbyderunderstøttelse af OpenCode AI-kodnings-IDE -- 🔗**TRAE-integration**— Fuld understøttelse af TRAE AI-udviklingsrammen -- 📦**Batch API**— Asynkron batchbehandling til masseanmodninger -- 🎯**Tag-baseret Routing**— Ruteanmodninger baseret på tilpassede tags og metadata -- 💰**Laveste omkostningsstrategi**— Vælg automatisk den billigste tilgængelige udbyder +### 🔜 Coming Soon -> 📝 Fuld funktionsspecifikationer tilgængelige i [`docs/new-features/`](docs/new-features/) (217 detaljerede specifikationer)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1970,18 +2245,20 @@ OmniRoute har**210+ funktioner planlagt**på tværs af flere udviklingsfaser. He ### How to Contribute -1. Fork depotet -2. Opret din feature-gren (`git checkout -b feature/amazing-feature`) -3. Bekræft dine ændringer (`git commit -m 'Tilføj fantastisk funktion'`) -4. Skub til grenen ("git push origin feature/amazing-feature") -5. Åbn en pull-anmodning +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Se [CONTRIBUTING.md](CONTRIBUTING.md) for detaljerede retningslinjer.### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1993,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Særlig tak til**[9router](https://github.com/decolua/9router)**af**[decolua](https://github.com/decolua)**— det originale projekt, der inspirerede denne gaffel. OmniRoute bygger på det utrolige fundament med yderligere funktioner, multimodale API'er og en fuld TypeScript-omskrivning. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Særlig tak til**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— den originale Go-implementering, der inspirerede denne JavaScript-port.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Licens -MIT-licens - se [LICENS](LICENS) for detaljer.--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/da/docs/ARCHITECTURE.md b/docs/i18n/da/docs/ARCHITECTURE.md index 223d04e314..5713fd9cd6 100644 --- a/docs/i18n/da/docs/ARCHITECTURE.md +++ b/docs/i18n/da/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Sidst opdateret: 2026-03-28_## Executive Summary -OmniRoute er en lokal AI-routinggateway og dashboard bygget på Next.js. -Det giver et enkelt OpenAI-kompatibelt slutpunkt (`/v1/*`) og dirigerer trafik på tværs af flere upstream-udbydere med oversættelse, fallback, token-opdatering og brugssporing. -Kerneegenskaber: +_Last updated: 2026-03-28_ -- OpenAI-kompatibel API-overflade til CLI/værktøjer (28 udbydere) -- Anmodning/svar oversættelse på tværs af udbyderformater -- Model combo fallback (multi-model sekvens) -- Fallback på kontoniveau (multi-konto pr. udbyder) -- Administration af forbindelse til OAuth + API-nøgleudbyder -- Indlejringsgenerering via `/v1/embeddings` (6 udbydere, 9 modeller) -- Billedgenerering via `/v1/images/generations` (4 udbydere, 9 modeller) -- Tænk tag-parsing (`...`) for ræsonneringsmodeller -- Response sanitization for streng OpenAI SDK-kompatibilitet -- Rollenormalisering (udvikler→system, system→bruger) for kompatibilitet på tværs af udbydere -- Struktureret outputkonvertering (json_schema → Gemini responseSchema) -- Lokal persistens for udbydere, nøgler, aliaser, kombinationer, indstillinger, priser -- Brug/omkostningssporing og anmodningslogning -- Valgfri skysynkronisering til synkronisering af flere enheder/tilstande -- IP-tilladelsesliste/blokeringsliste til API-adgangskontrol -- Tænkende budgetstyring (passthrough/auto/custom/adaptive) -- Global system prompt injektion -- Sessionssporing og fingeraftryk -- Forbedret prisbegrænsning pr. konto med udbyderspecifikke profiler -- Circuit breaker mønster for udbyderens modstandsdygtighed -- Anti-tordenbeskyttelse med mutex-låsning -- Signaturbaseret anmodnings deduplikeringscache -- Domænelag: modeltilgængelighed, omkostningsregler, fallback-politik, lockout-politik -- Vedvarende domænetilstand (SQLite-gennemskrivningscache til fallbacks, budgetter, lockouts, strømafbrydere) -- Politikmotor til centraliseret anmodningsevaluering (lockout → budget → fallback) -- Anmod om telemetri med p50/p95/p99 latency aggregering -- Korrelations-ID (X-Request-Id) til ende-til-ende-sporing -- Overholdelsesrevisionslogning med opt-out pr. API-nøgle -- Evalueringsramme for LLM kvalitetssikring -- Resilience UI-dashboard med strømafbryderstatus i realtid -- Modulære OAuth-udbydere (12 individuelle moduler under `src/lib/oauth/providers/`) +## Executive Summary -Primær runtime model: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- Next.js app-ruter under `src/app/api/*` implementerer både dashboard-API'er og kompatibilitets-API'er -- En delt SSE/routingkerne i `src/sse/*` + `open-sse/*` håndterer udbyderens udførelse, oversættelse, streaming, fallback og brug## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Lokal gateway køretid -- Dashboard management API'er -- Udbydergodkendelse og tokenopdatering -- Anmod om oversættelse og SSE-streaming -- Lokal stat + vedvarende brug -- Valgfri skysynkroniseringsorkestrering### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Implementering af skytjenester bag `NEXT_PUBLIC_CLOUD_URL` -- Udbyder SLA/kontrolplan uden for lokal proces -- Eksterne CLI-binære filer selv (Claude CLI, Codex CLI osv.)## Dashboard Surface (Current) +### Out of Scope -Hovedsider under `src/app/(dashboard)/dashboard/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — hurtig start + udbyderoversigt -- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint faner -- `/dashboard/providers` — udbyderforbindelser og legitimationsoplysninger -- `/dashboard/combos` — kombinationsstrategier, skabeloner, modelrutingsregler -- `/dashboard/costs` — prissammenlægning og prissynlighed -- `/dashboard/analytics` — brugsanalyse og -evalueringer -- `/dashboard/limits` — kvote-/satskontrol +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls - `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation -- `/dashboard/agents` — opdagede ACP-agenter + tilpasset agentregistrering -- `/dashboard/media` — billed-/video-/musiklegeplads -- `/dashboard/search-tools` — test af søgeudbydere og historik -- `/dashboard/health` — oppetid, strømafbrydere, hastighedsgrænser +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits - `/dashboard/logs` — request/proxy/audit/console logs -- `/dashboard/indstillinger` — systemindstillinger faner (generelt, routing, kombinationsstandarder osv.) -- `/dashboard/api-manager` — API-nøglelivscyklus og modeltilladelser## High-Level System Context +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Hovedmapper: +Main directories: -- `src/app/api/v1/*` og `src/app/api/v1beta/*` til kompatibilitets-API'er -- `src/app/api/*` til administrations-/konfigurations-API'er -- Næste omskrivninger i `next.config.mjs` map `/v1/*` til `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Vigtige kompatibilitetsruter: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` – inkluderer brugerdefinerede modeller med `custom: true` -- `src/app/api/v1/embeddings/route.ts` — indlejringsgenerering (6 udbydere) -- `src/app/api/v1/images/generations/route.ts` — billedgenerering (4+ udbydere inkl. Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedikeret chat pr. udbyder -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedikerede indlejringer pr. udbyder -- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedikerede billeder pr. udbyder +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` - `src/app/api/v1beta/models/[...path]/route.ts` -Ledelsesdomæner: +Management domains: -- Godkendelse/indstillinger: `src/app/api/auth/*`, `src/app/api/settings/*` -- Udbydere/forbindelser: `src/app/api/providers*` -- Provider noder: `src/app/api/provider-nodes*` -- Brugerdefinerede modeller: `src/app/api/provider-models` (GET/POST/DELETE) -- Modelkatalog: `src/app/api/models/route.ts` (GET) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) - Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` - Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- Brug: `src/app/api/usage/*` +- Usage: `src/app/api/usage/*` - Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` -- CLI-værktøjshjælpere: `src/app/api/cli-tools/*` -- IP-filter: `src/app/api/settings/ip-filter` (GET/PUT) -- Tænkebudget: `src/app/api/settings/thinking-budget` (GET/PUT) -- Systemprompt: `src/app/api/settings/system-prompt` (GET/PUT) -- Sessioner: `src/app/api/sessions` (GET) -- Satsgrænser: `src/app/api/rate-limits` (GET) -- Resiliens: `src/app/api/resilience` (GET/PATCH) — udbyderprofiler, afbryder, hastighedsgrænsetilstand -- Resilience reset: `src/app/api/resilience/reset` (POST) — nulstil breakers + cooldowns -- Cachestatistik: `src/app/api/cache/stats` (GET/DELETE) -- Modeltilgængelighed: `src/app/api/models/availability` (GET/POST) -- Telemetri: `src/app/api/telemetry/summary` (GET) +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) - Budget: `src/app/api/usage/budget` (GET/POST) -- Fallback-kæder: `src/app/api/fallback/chains` (GET/POST/DELETE) -- Overholdelsesrevision: `src/app/api/compliance/audit-log` (GET) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) - Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- Politikker: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Policies: `src/app/api/policies` (GET/POST) -Hovedflowmoduler: +## 2) SSE + Translation Core -- Indtastning: `src/sse/handlers/chat.ts` -- Kerneorkestrering: `open-sse/handlers/chatCore.ts` -- Udbyder eksekveringsadaptere: `open-sse/executors/*` -- Formatdetektion/udbyderkonfiguration: `open-sse/services/provider.ts` +Main flow modules: + +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` - Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Konto fallback logik: `open-sse/services/accountFallback.ts` -- Oversættelsesregister: `open-sse/translator/index.ts` -- Stream transformationer: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- Brugsudtrækning/normalisering: `open-sse/utils/usageTracking.ts` -- Tænk tag-parser: `open-sse/utils/thinkTagParser.ts` -- Indlejringshåndtering: `open-sse/handlers/embeddings.ts` -- Indlejringsudbyderregistrering: `open-sse/config/embeddingRegistry.ts` -- Billedgenereringshåndtering: `open-sse/handlers/imageGeneration.ts` -- Billedudbyderregistrering: `open-sse/config/imageRegistry.ts` -- Reaktionssanering: `open-sse/handlers/responseSanitizer.ts` -- Rollenormalisering: `open-sse/services/roleNormalizer.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -Tjenester (forretningslogik): +Services (business logic): -- Kontovalg/scoring: `open-sse/services/accountSelector.ts` +- Account selection/scoring: `open-sse/services/accountSelector.ts` - Context lifecycle management: `open-sse/services/contextManager.ts` -- Håndhævelse af IP-filter: `open-sse/services/ipFilter.ts` -- Sessionssporing: `open-sse/services/sessionManager.ts` -- Anmod om deduplikering: `open-sse/services/signatureCache.ts` -- Injektion af systemprompt: `open-sse/services/systemPrompt.ts` -- Tænkende budgetstyring: `open-sse/services/thinkingBudget.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` - Wildcard model routing: `open-sse/services/wildcardRouter.ts` -- Satsgrænsestyring: `open-sse/services/rateLimitManager.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` - Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -Domænelagsmoduler: +Domain layer modules: -- Modeltilgængelighed: `src/lib/domain/modelAvailability.ts` -- Omkostningsregler/budgetter: `src/lib/domain/costRules.ts` -- Fallback-politik: `src/lib/domain/fallbackPolicy.ts` +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` - Combo resolver: `src/lib/domain/comboResolver.ts` -- Lockout-politik: `src/lib/domain/lockoutPolicy.ts` -- Politikmotor: `src/domain/policyEngine.ts` — centraliseret lockout → budget → fallback-evaluering -- Fejlkodekatalog: `src/lib/domain/errorCodes.ts` -- Anmodnings-id: `src/lib/domain/requestId.ts` -- Hente timeout: `src/lib/domain/fetchTimeout.ts` -- Anmod om telemetri: `src/lib/domain/requestTelemetry.ts` -- Overholdelse/revision: `src/lib/domain/compliance/index.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` - Eval runner: `src/lib/domain/evalRunner.ts` -- Vedvarende domænetilstand: `src/lib/db/domainState.ts` — SQLite CRUD til reservekæder, budgetter, omkostningshistorik, lockouttilstand, strømafbrydere +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -OAuth-udbydermoduler (12 individuelle filer under `src/lib/oauth/providers/`): +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -- Registerindeks: `src/lib/oauth/providers/index.ts` -- Individuelle udbydere: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts.line`, `t.`s.`s.`s.` -- Tyndt omslag: `src/lib/oauth/providers.ts` — reeksporterer fra individuelle moduler## 3) Persistence Layer +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -Primær tilstand DB (SQLite): +## 3) Persistence Layer -- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrationer, WAL) -- Re-eksport facade: `src/lib/localDb.ts` (tyndt kompatibilitetslag for opkaldere) -- fil: `${DATA_DIR}/storage.sqlite` (eller `$XDG_CONFIG_HOME/omniroute/storage.sqlite` når indstillet, ellers `~/.omniroute/storage.sqlite`) -- enheder (tabeller + KV-navnerum): providerConnections, providerNodes, modelAliaser, combos, apiKeys, indstillinger, prissætning,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +Primary state DB (SQLite): -Brugsvedholdenhed: +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -- facade: `src/lib/usageDb.ts` (dekomponerede moduler i `src/lib/usage/*`) -- SQLite-tabeller i `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` -- valgfrie filartefakter forbliver for kompatibilitet/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- Ældre JSON-filer migreres til SQLite ved opstartsmigreringer, når de er til stede +Usage persistence: + +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present Domain State DB (SQLite): -- `src/lib/db/domainState.ts` — CRUD-operationer for domænetilstand -- Tabeller (oprettet i `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- Gennemskrivningscachemønster: Kort i hukommelsen er autoritative under kørsel; mutationer skrives synkront til SQLite; tilstand gendannes fra DB ved koldstart## 4) Auth + Security Surfaces +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces - Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- Generering/bekræftelse af API-nøgler: `src/shared/utils/apiKey.ts` -- Udbyderhemmeligheder vedblev i 'providerConnections'-poster -- Udgående proxy-understøttelse via `open-sse/utils/proxyFetch.ts` (env vars) og `open-sse/utils/networkProxy.ts` (konfigurerbar pr. udbyder eller global)## 5) Cloud Sync +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync - Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Periodisk opgave: `src/shared/services/cloudSyncScheduler.ts` -- Periodisk opgave: `src/shared/services/modelSyncScheduler.ts` -- Styr rute: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Fallback-beslutninger er drevet af `open-sse/services/accountFallback.ts` ved hjælp af statuskoder og fejlmeddelelsesheuristik. Combo-routing tilføjer en ekstra beskyttelse: udbyder-omfattede 400'er, såsom upstream-indholdsblokering og rollevalideringsfejl, behandles som model-lokale fejl, så senere combo-mål kan stadig køre.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -Opdatering under live trafik udføres inde i `open-sse/handlers/chatCore.ts` via eksekveren `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -Periodisk synkronisering udløses af "CloudSyncScheduler", når skyen er aktiveret.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -Fysiske lagerfiler: +Physical storage files: -- primær runtime DB: `${DATA_DIR}/storage.sqlite` -- anmode om log linjer: `${DATA_DIR}/log.txt` (compat/debug artefakt) -- strukturerede opkaldsdataarkiver: `${DATA_DIR}/call_logs/` -- valgfri oversætter/anmodningsfejlfindingssessioner: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: kompatibilitets-API'er -- `src/app/api/v1/providers/[provider]/*`: dedikerede ruter pr. udbyder (chat, indlejringer, billeder) -- `src/app/api/providers*`: udbyder CRUD, validering, test -- `src/app/api/provider-nodes*`: tilpasset kompatibel nodestyring -- `src/app/api/provider-models`: Custom model management (CRUD) -- `src/app/api/models/route.ts`: modelkatalog API (aliaser + tilpassede modeller) -- `src/app/api/oauth/*`: OAuth/enhedskode-flow -- `src/app/api/keys*`: lokal API-nøglelivscyklus -- `src/app/api/models/alias`: aliashåndtering +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management - `src/app/api/combos*`: fallback combo management -- `src/app/api/pricing`: pristilsidesættelser til omkostningsberegning -- `src/app/api/settings/proxy`: proxy-konfiguration (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: test af udgående proxyforbindelse (POST) -- `src/app/api/usage/*`: brugs- og log-API'er -- `src/app/api/sync/*` + `src/app/api/cloud/*`: skysynkronisering og skyvendte hjælpere -- `src/app/api/cli-tools/*`: lokale CLI-konfigurationsskrivere/checkers -- `src/app/api/settings/ip-filter`: IP-tilladelsesliste/blokeringsliste (GET/PUT) -- `src/app/api/settings/thinking-budget`: Tænketoken-budgetkonfiguration (GET/PUT) -- `src/app/api/settings/system-prompt`: global systemprompt (GET/PUT) -- `src/app/api/sessions`: aktiv sessionsfortegnelse (GET) -- `src/app/api/rate-limits`: rategrænsestatus pr. konto (GET)### Routing and Execution Core +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: anmodning om parse, kombinationshåndtering, kontovalgsløkke -- `open-sse/handlers/chatCore.ts`: oversættelse, eksekutorafsendelse, genforsøg/opdateringshåndtering, stream-opsætning -- `open-sse/executors/*`: udbyderspecifik netværks- og formatadfærd### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: oversætterregister og orkestrering -- Anmod om oversættere: `open-sse/translator/request/*` -- Svaroversættere: `open-sse/translator/response/*` -- Formatkonstanter: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: persistent config/state og domæne persistens på SQLite -- `src/lib/localDb.ts`: re-eksport af kompatibilitet til DB-moduler -- `src/lib/usageDb.ts`: brugshistorik/opkaldslogs facade oven på SQLite-tabeller## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Hver udbyder har en specialiseret executor, der udvider `BaseExecutor` (i `open-sse/executors/base.ts`), som giver URL-opbygning, header-konstruktion, genforsøg med eksponentiel backoff, credential refresh hooks og `execute()`-orkestreringsmetoden. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Eksekutør | Udbyder(e) | Særlig håndtering | -| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------- | -| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fyrværkeri, Cerebras, Cohere, NVIDIA | Dynamisk URL/header-konfiguration pr. udbyder | -| `AntigravityExecutor` | Google Antigravity | Brugerdefinerede projekt-/sessions-id'er, forsøg igen - efter parsing | -| `CodexExecutor` | OpenAI Codex | Injicerer systeminstruktioner, fremtvinger ræsonnement indsats | -| `CursorExecutor` | Markør IDE | ConnectRPC-protokol, Protobuf-kodning, anmodningssignering via checksum | -| `GithubExecutor` | GitHub Copilot | Copilot token opdatering, VSCode-mimicing headers | -| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binært format → SSE-konvertering | -| `GeminiCLIEexecutor` | Gemini CLI | Opdateringscyklus for Google OAuth-token | +### Persistence -Alle andre udbydere (inklusive brugerdefinerede kompatible noder) bruger `DefaultExecutor`.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Udbyder | Format | Auth | Stream | Ikke-stream | Token Opdater | Brug API | -| ---------------- | --------------- | --------------------- | ---------------- | ----------- | ------------- | -------------------- | ------------------------------ | -| Claude | claude | API-nøgle / OAuth | ✅ | ✅ | ✅ | ⚠️ Kun administrator | -| Tvillingerne | gemini | API-nøgle / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | -| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | -| Antigravitation | antityngdekraft | OAuth | ✅ | ✅ | ✅ | ✅ Fuld kvote API | -| OpenAI | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ | -| Codex | openai-svar | OAuth | ✅ tvunget | ❌ | ✅ | ✅ Satsgrænser | -| GitHub Copilot | åbne | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Kvote snapshots | -| Markør | markør | Tilpasset kontrolsum | ✅ | ✅ | ❌ | ❌ | -| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Brugsgrænser | -| Qwen | åbne | OAuth | ✅ | ✅ | ✅ | ⚠️ Efter anmodning | -| Qoder | åbne | OAuth (Grundlæggende) | ✅ | ✅ | ✅ | ⚠️ Efter anmodning | -| OpenRouter | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | claude | API-nøgle | ✅ | ✅ | ❌ | ❌ | -| DeepSeek | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ | -| Groq | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ | -| Mistral | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ | -| Forvirring | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ | -| Sammen AI | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ | -| Fyrværkeri AI | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ | -| Cerebras | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ | -| Sammenhæng | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Detekterede kildeformater omfatter: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. -- 'openai' -- `openai-svar` -- `Claude` -- 'tvilling' +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | -Målformater omfatter: +All other providers (including custom compatible nodes) use the `DefaultExecutor`. -- OpenAI chat/svar +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: + +- `openai` +- `openai-responses` +- `claude` +- `gemini` + +Target formats include: + +- OpenAI chat/Responses - Claude -- Gemini/Gemini-CLI/Antigravity kuvert +- Gemini/Gemini-CLI/Antigravity envelope - Kiro -- Markør +- Cursor -Oversættelser bruger**OpenAI som hub-format**- alle konverteringer går gennem OpenAI som mellemliggende:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Oversættelser vælges dynamisk baseret på kildens nyttelastform og udbyderens målformat. +Additional processing layers in the translation pipeline: -Yderligere behandlingslag i oversættelsespipelinen: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Responssanering**— Fjerner ikke-standardfelter fra OpenAI-formatsvar (både streaming og ikke-streaming) for at sikre streng SDK-overholdelse --**Rollenormalisering**— Konverterer `udvikler` → `system` til ikke-OpenAI-mål; fletter `system` → `bruger` for modeller, der afviser systemrollen (GLM, ERNIE) --**Tænk tag-udtrækning**— Parser "..."-blokke fra indhold til feltet "reasoning_content" --**Structured output**— Konverterer OpenAI `response_format.json_schema` til Geminis `responseMimeType` + `responseSchema`## Supported API Endpoints +## Supported API Endpoints -| Slutpunkt | Format | Behandler | -| -------------------------------------------------- | ------------------ | -------------------------------------------------------------------------- | -| `POST /v1/chat/afslutninger` | OpenAI Chat | `src/sse/handlers/chat.ts` | -| `POST /v1/meddelelser` | Claude Beskeder | Samme handler (auto-detekteret) | -| `POST /v1/svar` | OpenAI-svar | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/indlejringer` | OpenAI-indlejringer | `open-sse/handlers/embeddings.ts` | -| `GET /v1/indlejringer` | Modelliste | API-rute | -| `POST /v1/billeder/generationer` | OpenAI Billeder | `open-sse/handlers/imageGeneration.ts` | -| `GET /v1/billeder/generationer` | Modelliste | API-rute | -| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedikeret per udbyder med modelvalidering | -| `POST /v1/providers/{provider}/embeddings` | OpenAI-indlejringer | Dedikeret per udbyder med modelvalidering | -| `POST /v1/providers/{provider}/images/generations` | OpenAI Billeder | Dedikeret per udbyder med modelvalidering | -| `POST /v1/messages/count_tokens` | Claude Token Count | API-rute | -| `GET /v1/modeller` | OpenAI Models liste | API-rute (chat + indlejring + billede + brugerdefinerede modeller) | -| `GET /api/models/catalog` | Katalog | Alle modeller grupperet efter udbyder + type | -| `POST /v1beta/models/*:streamGenerateContent` | Tvilling hjemmehørende | API-rute | -| `GET/PUT/DELETE /api/indstillinger/proxy` | Proxy-konfiguration | Netværk proxy-konfiguration | -| `POST /api/settings/proxy/test` | Proxy-forbindelse | Proxy-sundheds-/forbindelsestestslutpunkt | -| `GET/POST/DELETE /api/provider-models` | Udbyder modeller | Udbydermodelmetadata understøtter tilpassede og administrerede tilgængelige modeller |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -Bypass-handleren (`open-sse/utils/bypassHandler.ts`) opsnapper kendte "throwaway"-anmodninger fra Claude CLI - opvarmningsping, titeludtræk og tokentællinger - og returnerer et**falsk svar**uden at forbruge upstream-udbydertokens. Dette udløses kun, når `User-Agent` indeholder `claude-cli`.## Request Logger Pipeline +## Bypass Handler -Anmodningsloggeren (`open-sse/utils/requestLogger.ts`) giver en 7-trins debug-logningspipeline, deaktiveret som standard, aktiveret via `ENABLE_REQUEST_LOGS=true`:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Filer skrives til `/logs//` for hver anmodningssession.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- Nedkøling af udbyderkonto på forbigående/rate/godkendelsesfejl -- konto fallback før mislykket anmodning -- combo model fallback, når den nuværende model/udbydersti er udtømt## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- Forhåndstjek og opdater med genforsøg for udbydere, der kan opdateres -- 401/403 forsøg igen efter opdateringsforsøg i kernestien## 3) Stream Safety +## 2) Token Expiry -- afbrydelsesbevidst streamcontroller -- oversættelsesstrøm med slut-af-stream-skyl og `[DONE]`-håndtering -- forbrugsestimeret fallback, når udbyderens brugsmetadata mangler## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- Synkroniseringsfejl dukker op, men lokal kørsel fortsætter -- Scheduler har logik, der kan genforsøge, men periodisk udførelse kalder i øjeblikket enkelt-forsøgssynkronisering som standard## 5) Data Integrity +## 3) Stream Safety -- SQLite-skemamigreringer og auto-opgraderingshooks ved opstart -- ældre JSON → SQLite-migreringskompatibilitetssti## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Kilder til synlighed ved kørsel: +## 4) Cloud Sync Degradation -- konsollogfiler fra `src/sse/utils/logger.ts` -- brugsaggregater pr. anmodning i SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- fire-trins detaljeret nyttelastfangst i SQLite (`request_detail_logs`), når `settings.detailed_logs_enabled=true` -- statuslog for tekstanmodning i `log.txt` (valgfrit/kompatibelt) -- valgfri dybe anmodnings-/oversættelseslogfiler under `logs/` når `ENABLE_REQUEST_LOGS=true` -- dashboardbrugsslutpunkter (`/api/usage/*`) for brugergrænsefladeforbrug +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -Detaljeret anmodning om nyttelastfangst gemmer op til fire JSON-nyttelasttrin pr. dirigeret opkald: +## 5) Data Integrity -- rå anmodning modtaget fra klienten -- oversat anmodning faktisk sendt opstrøms -- Providersvar rekonstrueret som JSON; streamede svar komprimeres til den endelige oversigt plus stream metadata -- endelig kundesvar returneret af OmniRoute; streamede svar gemmes i den samme kompakte oversigtsform## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- JWT-hemmelighed (`JWT_SECRET`) sikrer bekræftelse/signering af dashboard-sessionscookie -- Oprindelig adgangskode-bootstrap ('INITIAL_PASSWORD') skal eksplicit konfigureres til førstegangs-klargøring -- API-nøgle HMAC-hemmelighed (`API_KEY_SECRET`) sikrer genereret lokalt API-nøgleformat -- Udbyderhemmeligheder (API-nøgler/tokens) bevares i lokal DB og bør beskyttes på filsystemniveau -- Slutpunkter for skysynkronisering er afhængige af API-nøglegodkendelse + maskin-id-semantik## Environment and Runtime Matrix +## Observability and Operational Signals -Miljøvariabler aktivt brugt af kode: +Runtime visibility sources: -- App/godkendelse: `JWT_SECRET`, `INITIAL_PASSWORD` -- Lager: `DATA_DIR` -- Kompatibel nodeadfærd: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Valgfri lagerbasetilsidesættelse (Linux/macOS, når `DATA_DIR` er deaktiveret): `XDG_CONFIG_HOME` -- Sikkerhedshashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Logning: `ENABLE_REQUEST_LOGS` -- Synkronisering/sky-URL: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- Udgående proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` og varianter med små bogstaver -- SOCKS5-funktionsflag: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- Platform/runtime-hjælpere (ikke app-specifik konfiguration): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` og `localDb` deler den samme basismappepolitik (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) med ældre filmigrering. -2. `/api/v1/route.ts` uddelegerer til den samme forenede katalogbygger, der bruges af `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) for at undgå semantisk drift. -3. Anmodningslogger skriver hele headers/body, når den er aktiveret; behandle logbiblioteket som følsomt. -4. Cloudadfærd afhænger af korrekt `NEXT_PUBLIC_BASE_URL` og cloud-endepunkts tilgængelighed. -5. `open-sse/` biblioteket udgives som `@omniroute/open-sse`**npm workspace-pakken**. Kildekoden importerer den via `@omniroute/open-sse/...` (løst af Next.js `transpilePackages`). Filstier i dette dokument bruger stadig mappenavnet `open-sse/` for at opnå konsistens. -6. Diagrammer i dashboardet bruger**Recharts**(SVG-baseret) til tilgængelige, interaktive analysevisualiseringer (søjlediagrammer for modelbrug, udbyderopdelingstabeller med succesrater). -7. E2E-tests bruger**Playwright**(`tests/e2e/`), køres via `npm run test:e2e`. Enhedstests bruger**Node.js test runner**(`tests/unit/`), køres via `npm run test:unit`. Kildekoden under `src/` er**TypeScript**(`.ts`/`.tsx`); `open-sse/`-arbejdsområdet forbliver JavaScript (`.js`). -8. Siden Indstillinger er organiseret i 5 faner: Sikkerhed, Routing (6 globale strategier: fill-first, round-robin, p2c, random, mindst brugt, omkostningsoptimeret), Resiliens (redigerbare hastighedsgrænser, strømafbryder, politikker), AI (tænkebudget, systemprompt, promptcache), Avanceret (proxy).## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- Byg fra kilde: `npm run build` -- Byg Docker-billede: `docker build -t omniroute .` -- Start service og bekræft: -- `GET /api/indstillinger` -- `GET /api/v1/modeller` -- CLI-målbasis-URL skal være "http://:20128/v1", når "PORT=20128" +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` +- `GET /api/v1/models` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/da/docs/FEATURES.md b/docs/i18n/da/docs/FEATURES.md index 19884147f8..7c798a15dd 100644 --- a/docs/i18n/da/docs/FEATURES.md +++ b/docs/i18n/da/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Visuel guide til hver sektion af OmniRoute-dashboardet.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Administrer AI-udbyderforbindelser: OAuth-udbydere (Claude Code, Codex, Gemini CLI), API-nøgleudbydere (Groq, DeepSeek, OpenRouter) og gratis udbydere (Qoder, Qwen, Kiro). Kiro-konti inkluderer sporing af kreditsaldo - resterende kreditter, samlet godtgørelse og fornyelsesdato synlig i Dashboard → Brug.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Opret modelrouting-kombinationer med 6 strategier: prioritet, vægtet, round-robin, tilfældig, mindst brugt og omkostningsoptimeret. Hver combo kæder flere modeller med automatisk fallback og inkluderer hurtige skabeloner og klarhedstjek.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Omfattende brugsanalyse med token-forbrug, omkostningsestimater, aktivitetsvarmekort, ugentlige distributionsdiagrammer og opdelinger pr. udbyder.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Overvågning i realtid: oppetid, hukommelse, version, latency percentiler (p50/p95/p99), cache-statistik og udbyderens afbrydertilstande.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Fire tilstande til fejlfinding af API-oversættelser:**Playground**(formatkonverter),**Chat Tester**(live-anmodninger),**Test Bench**(batchtest) og**Live Monitor**(streaming i realtid).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Test enhver model direkte fra instrumentbrættet. Vælg udbyder, model og slutpunkt, skriv prompts med Monaco Editor, stream svar i realtid, afbryd midt-stream, og se timing-metrics.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Brugerdefinerbare farvetemaer til hele dashboardet. Vælg mellem 7 forudindstillede farver (koral, blå, rød, grøn, violet, orange, cyan) eller opret et brugerdefineret tema ved at vælge en hex-farve. Understøtter lys, mørk og systemtilstand.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Omfattende indstillingspanel med faner: +Comprehensive settings panel with tabs: --**Generelt**— Systemlagring, backupstyring (eksport/importdatabase) -**Udseende**— Temavælger (mørke/lys/system), forudindstillinger af farvetema og brugerdefinerede farver, synlighed i sundhedslog, synlighedskontrol for sidebjælkeelementer -**Sikkerhed**— API-endepunktsbeskyttelse, tilpasset udbyderblokering, IP-filtrering, sessionsoplysninger -**Routing**— Modelaliaser, forringelse af baggrundsopgaver -**Resiliens**— Frekvensgrænsevedholdenhed, tuning af strømafbryder, automatisk deaktivering af forbudte konti, overvågning af udbyderens udløb -**Avanceret**— Konfigurationstilsidesættelser, konfigurationsrevisionsspor, fallback-forringelsestilstand![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Et-klik-konfiguration til AI-kodningsværktøjer: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor og Factory Droid. Indeholder automatiseret konfigurationsanvendelse/nulstilling, forbindelsesprofiler og modelkortlægning.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Dashboard til at opdage og administrere CLI-agenter. Viser et gitter med 14 indbyggede agenter (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) med: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Installationsstatus**— Installeret / Ikke fundet med versionsregistrering -**Protokolmærker**— stdio, HTTP osv. -**Tilpassede agenter**- Registrer ethvert CLI-værktøj via formular (navn, binær, versionskommando, spawn args) -**CLI Fingerprint Matching**— Skift pr. udbyder for at matche native CLI-anmodningssignaturer, hvilket reducerer risikoen for forbud, mens proxy-IP bevares--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Generer billeder, videoer og musik fra dashboardet. Understøtter OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open og MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Anmodningslogning i realtid med filtrering efter udbyder, model, konto og API-nøgle. Viser statuskoder, tokenbrug, latenstid og svardetaljer.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Dit forenede API-slutpunkt med kapacitetsopdeling: Chatfuldførelser, Responses API, indlejringer, billedgenerering, omrangering, lydtransskription, tekst-til-tale, modereringer og registrerede API-nøgler. Cloudflare Quick Tunnel integration og cloud proxy support til fjernadgang.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -Opret, omfang og tilbagekald API-nøgler. Hver nøgle kan begrænses til specifikke modeller/udbydere med fuld adgang eller skrivebeskyttet tilladelse. Visuel nøglestyring med brugssporing.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Administrativ handlingssporing med filtrering efter handlingstype, aktør, mål, IP-adresse og tidsstempel. Fuld historik for sikkerhedshændelser.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Native Electron desktop-app til Windows, macOS og Linux. Kør OmniRoute som et selvstændigt program med systembakkeintegration, offline support, automatisk opdatering og installation med ét klik. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Nøglefunktioner: +Key features: -- Afstemning af serverberedskab (ingen tom skærm ved koldstart) -- Systembakke med portstyring -- Indholdssikkerhedspolitik -- Engangslås -- Automatisk opdatering ved genstart -- Platform-betinget UI (macOS trafiklys, Windows/Linux standard titellinje) -- Hærdet Electron build-emballage — symlinkede 'node_modules' i den selvstændige bundt detekteres og afvises før pakning, hvilket forhindrer runtime-afhængighed af build-maskinen (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Se [`electron/README.md`](../electron/README.md) for fuld dokumentation. +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/da/docs/TROUBLESHOOTING.md b/docs/i18n/da/docs/TROUBLESHOOTING.md index 67794ed8eb..063677c844 100644 --- a/docs/i18n/da/docs/TROUBLESHOOTING.md +++ b/docs/i18n/da/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -Almindelige problemer og løsninger til OmniRoute.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Problem | Løsning | -| -------------------------------------- | --------------------------------------------------------------------------- | --- | -| Første login virker ikke | Indstil `INITIAL_PASSWORD` i `.env` (ingen hardcoded standard) | -| Dashboard åbner ved forkert port | Indstil `PORT=20128` og `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Ingen anmodningslogfiler under `logs/` | Indstil `ENABLE_REQUEST_LOGS=true` | -| EACCES: tilladelse nægtet | Indstil `DATA_DIR=/path/to/writable/dir` for at tilsidesætte `~/.omniroute` | -| Routingstrategi gemmer ikke | Opdatering til v1.4.11+ (Zod-skemafix for indstillinger persistens) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Årsag:**Udbyderkvoten er opbrugt. +**Cause:** Provider quota exhausted. -**Ret:** +**Fix:** -1. Tjek dashboard-kvotesporing -2. Brug en kombination med reserveniveauer -3. Skift til billigere/gratis niveau### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Årsag:**Abonnementskvoten er opbrugt. +### Rate Limiting -**Ret:** +**Cause:** Subscription quota exhausted. -- Tilføj reserve: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Brug GLM/MiniMax som billig backup### OAuth Token Expired +**Fix:** -OmniRoute opdaterer automatisk tokens. Hvis problemerne fortsætter: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Dashboard → Udbyder → Genopret forbindelse -2. Slet og tilføj udbyderforbindelsen igen--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Bekræft, at `BASE_URL` peger på din kørende forekomst (f.eks. `http://localhost:20128`) -2. Bekræft "CLOUD_URL" peger på dit cloud-slutpunkt (f.eks. "https://omniroute.dev") -3. Hold `NEXT_PUBLIC_*`-værdier på linje med værdier på serversiden### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Symptom:**`Uventet token 'd'...` på cloud-slutpunktet for ikke-streaming-opkald. +### Cloud `stream=false` Returns 500 -**Årsag:**Upstream returnerer SSE-nyttelast, mens klienten forventer JSON. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Løsning:**Brug 'stream=true' til direkte skyopkald. Lokal kørselstid inkluderer SSE→JSON fallback.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Opret en ny nøgle fra det lokale dashboard (`/api/keys`) -2. Kør skysynkronisering: Aktiver sky → Synkroniser nu -3. Gamle/ikke-synkroniserede nøgler kan stadig returnere '401' på skyen--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Tjek runtime-felter: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. For bærbar tilstand: brug billedmål "runner-cli" (bundtet CLI'er) -3. For værtsmonteringstilstand: indstil `CLI_EXTRA_PATHS` og monter host bin-mappe som skrivebeskyttet -4. Hvis `installed=true` og `runnable=false`: binær blev fundet, men sundhedstjekket mislykkedes### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Tjek brugsstatistik i Dashboard → Brug -2. Skift primær model til GLM/MiniMax -3. Brug gratis niveau (Gemini CLI, Qoder) til ikke-kritiske opgaver -4. Indstil omkostningsbudgetter pr. API-nøgle: Dashboard → API-nøgler → Budget--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Indstil `ENABLE_REQUEST_LOGS=true` i din `.env`-fil. Logs vises under mappen `logs/`.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Hovedtilstand: `${DATA_DIR}/storage.sqlite` (udbydere, kombinationer, aliaser, nøgler, indstillinger) -- Anvendelse: SQLite-tabeller i `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + valgfri `${DATA_DIR}/log.txt` og `${DATA_DIR}/call_logs/` -- Anmodningslogfiler: `/logs/...` (når `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Når en udbyders afbryder er ÅBEN, blokeres anmodninger, indtil nedkølingen udløber. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Ret:** +**Fix:** -1. Gå til**Dashboard → Indstillinger → Resiliens** -2. Tjek afbryderkortet for den berørte udbyder -3. Klik på**Nulstil alle**for at rydde alle afbrydere, eller vent på, at nedkølingen udløber -4. Bekræft, at udbyderen faktisk er tilgængelig, før du nulstiller### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Hvis en udbyder gentagne gange går i ÅBEN tilstand: +### Provider keeps tripping the circuit breaker -1. Tjek**Dashboard → Health → Provider Health**for fejlmønsteret -2. Gå til**Indstillinger → Resiliens → Udbyderprofiler**og øg fejltærsklen -3. Tjek, om udbyderen har ændret API-grænser eller kræver gengodkendelse -4. Gennemgå latency-telemetri — høj latenstid kan forårsage timeout-baserede fejl--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Sørg for, at du bruger det korrekte præfiks: `deepgram/nova-3` eller `assemblyai/best` -- Bekræft, at udbyderen er tilsluttet i**Dashboard → Udbydere**### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Tjek understøttede lydformater: "mp3", "wav", "m4a", "flac", "ogg", "webm" -- Bekræft filstørrelsen er inden for udbyderens grænser (typisk < 25 MB) -- Tjek gyldigheden af udbyderens API-nøgle på udbyderkortet--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Brug**Dashboard → Oversætter**til at fejlfinde problemer med formatoversættelse: +Use **Dashboard → Translator** to debug format translation issues: -| Tilstand | Hvornår skal man bruge | -| ---------------- | --------------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Legeplads** | Sammenlign input/output-formater side om side — indsæt en mislykket anmodning for at se, hvordan den oversættes | -| **Chattester** | Send livebeskeder og inspicer den fulde anmodnings-/svarnyttelast inklusive overskrifter | -| **Testbænk** | Kør batchtest på tværs af formatkombinationer for at finde ud af, hvilke oversættelser der er brudte | -| **Live Monitor** | Se anmodningsflow i realtid for at fange periodiske oversættelsesproblemer | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Tænke-tags vises ikke**— Tjek, om måludbyderen understøtter tænkning og indstilling af tænkebudget -**Værktøjsopkald falder**— Nogle formatoversættelser kan fjerne ikke-understøttede felter; verificere i Playground-tilstand -**Systemprompt mangler**— Claude og Gemini håndterer systemprompts forskelligt; kontrollere oversættelsesoutput -**SDK returnerer rå streng i stedet for objekt**— Rettet i v1.1.0: svar sanitizer fjerner nu ikke-standard felter (`x_groq`, `usage_breakdown` osv.), der forårsager OpenAI SDK Pydantic valideringsfejl -**GLM/ERNIE afviser 'system'-rolle**— Rettet i v1.1.0: Rollenormalisering flettes automatisk systemmeddelelser ind i brugermeddelelser for inkompatible modeller -**"udvikler"-rolle ikke genkendt**- Rettet i v1.1.0: automatisk konverteret til "system" for ikke-OpenAI-udbydere -**`json_schema` virker ikke med Gemini**- Rettet i v1.1.0: `response_format` er nu konverteret til Gemini's `responseMimeType` + `responseSchema`--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- Automatisk hastighedsgrænse gælder kun for API-nøgleudbydere (ikke OAuth/abonnement) -- Bekræft, at**Indstillinger → Modstandsdygtighed → Udbyderprofiler**har aktiveret automatisk satsgrænse -- Tjek, om udbyderen returnerer '429'-statuskoder eller 'Retry-After'-overskrifter### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -Udbyderprofiler understøtter disse indstillinger: +### Tuning exponential backoff --**Base delay**— Indledende ventetid efter første fejl (standard: 1s) -**Maksimal forsinkelse**— Maksimal ventetid (standard: 30s) -**Multiplikator**— Hvor meget skal forsinkelsen øges pr. på hinanden følgende fejl (standard: 2x)### Anti-thundering herd +Provider profiles support these settings: -Når mange samtidige anmodninger rammer en hastighedsbegrænset udbyder, bruger OmniRoute mutex + automatisk hastighedsbegrænsning til at serialisere anmodninger og forhindre kaskadefejl. Dette er automatisk for API-nøgleudbydere.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Nogle OmniRoute-brugere placerer gatewayen foran RAG- eller agentstakke. I disse opsætninger er det almindeligt at se et mærkeligt mønster: OmniRoute ser sund ud (udbydere op, routing profiler ok, ingen hastighedsgrænse advarsler), men det endelige svar er stadig forkert. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -I praksis kommer disse hændelser normalt fra RAG-rørledningen nedstrøms, ikke fra selve gatewayen. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Hvis du ønsker et fælles ordforråd til at beskrive disse fejl, kan du bruge WFGY ProblemMap, en ekstern MIT-licenstekstressource, der definerer seksten tilbagevendende RAG/LLM-fejlmønstre. På et højt niveau dækker det over: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- genfindingsdrift og brudte kontekstgrænser -- tomme eller uaktuelle indekser og vektorlagre -- indlejring versus semantisk mismatch -- problemer med hurtig montering og kontekstvindue -- logisk sammenbrud og oversikre svar -- lang kæde og agentkoordinationsfejl -- multiagent hukommelse og rolledrift -- problemer med implementering og bootstrap-bestilling +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -Ideen er enkel: +The idea is simple: -1. Når du undersøger et dårligt svar, skal du fange: - - brugeropgave og anmodning - - rute eller udbyderkombination i OmniRoute - - enhver RAG-kontekst, der bruges downstream (hentede dokumenter, værktøjsopkald osv.) -2. Kortlæg hændelsen til et eller to WFGY ProblemMap-numre (`No.1` … `No.16`). -3. Gem nummeret i dit eget dashboard, runbook eller hændelsessporing ved siden af ​​OmniRoute-logfilerne. -4. Brug den tilsvarende WFGY-side til at beslutte, om du skal ændre din RAG-stack, retriever eller routingstrategi. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -Fuld tekst og konkrete opskrifter live her (MIT-licens, kun tekst): +Full text and concrete recipes live here (MIT license, text only): [WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -Du kan ignorere dette afsnit, hvis du ikke kører RAG eller agentpipelines bag OmniRoute.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**GitHub-problemer**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Architecture**: Se [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for interne detaljer -**API-reference**: Se [`docs/API_REFERENCE.md`](API_REFERENCE.md) for alle endepunkter -**Health Dashboard**: Tjek**Dashboard → Health**for systemstatus i realtid -**Oversætter**: Brug**Dashboard → Oversætter**til at fejlsøge formatproblemer +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/da/llm.txt b/docs/i18n/da/llm.txt new file mode 100644 index 0000000000..e245fa1d2d --- /dev/null +++ b/docs/i18n/da/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Dansk) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Overblik + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Sikkerhed +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/de/README.md b/docs/i18n/de/README.md index 8a56a17378..3019d440b4 100644 --- a/docs/i18n/de/README.md +++ b/docs/i18n/de/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Ihr universeller API-Proxy – ein Endpunkt, über 60 Anbieter, keine Ausfallzeiten. Jetzt mit**MCP Server (25 Tools)**,**A2A-Protokoll**,**Speicher-/Skills-Systeme**und**Electron Desktop App**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Chat-Abschlüsse • Einbettungen • Bildgenerierung • Video • Musik • Audio • Reranking •**Websuche**• MCP-Server • A2A-Protokoll • 100 % TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Ihr universeller API-Proxy – ein Endpunkt, über 60 Anbieter, keine Ausfallze [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Website](https://omniroute.online) • [🚀 Schnellstart](#-quick-start) • [💡 Funktionen](#-key-features) • [📖 Dokumente](#-documentation) • [💰 Preise](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Verfügbar in:**🇺🇸 [Englisch](README.md) | 🇧🇷 [Português (Brasilien)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italienisch](docs/i18n/it/README.md) | 🇷🇺 [Russisch](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dänisch](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Niederlande](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polnisch](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,28 +60,30 @@ _Ihr universeller API-Proxy – ein Endpunkt, über 60 Anbieter, keine Ausfallze ## 📸 Dashboard Preview -
-Klicken Sie hier, um Dashboard-Screenshots anzuzeigen +
+Click to see dashboard screenshots -| Seite | Screenshot | -| ---------------------- | -------------------------------------------------- | ---------- | -| **Anbieter** | ![Anbieter](docs/screenshots/01-providers.png) | -| **Kombinationen** | ![Combos](docs/screenshots/02-combos.png) | -| **Analytik** | ![Analytics](docs/screenshots/03-analytics.png) | -| **Gesundheit** | ![Gesundheit](docs/screenshots/04-health.png) | -| **Übersetzer** | ![Übersetzer](docs/screenshots/05-translator.png) | -| **Einstellungen** | ![Einstellungen](docs/screenshots/06-settings.png) | -| **CLI-Tools** | ![CLI-Tools](docs/screenshots/07-cli-tools.png) | -| **Nutzungsprotokolle** | ![Nutzung](docs/screenshots/08-usage.png) | -| **Endpunkte** | ![Endpunkte](docs/screenshots/09-endpoint.png) |
| +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | + +
--- ### 🤖 Free AI Provider for your favorite coding agents -_Verbinden Sie jedes KI-gestützte IDE- oder CLI-Tool über OmniRoute – kostenloses API-Gateway für unbegrenzte Codierung._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ - + @@ -123,483 +132,557 @@ _Verbinden Sie jedes KI-gestützte IDE- oder CLI-Tool über OmniRoute – kosten
@@ -88,28 +97,28 @@ _Verbinden Sie jedes KI-gestützte IDE- oder CLI-Tool über OmniRoute – kosten NanoBot
NanoBot

- ⭐ 20,9K + ⭐ 20.9K
PicoClaw
PicoClaw

- ⭐ 14,6K + ⭐ 14.6K
ZeroClaw
ZeroClaw

- ⭐ 9,9K + ⭐ 9.9K
IronClaw
- Eisenklaue + IronClaw

- ⭐ 2,1K + ⭐ 2.1K
Codex CLI
- Codex-CLI + Codex CLI

- ⭐ 60,8K + ⭐ 60.8K
Claude Code
Claude Code

- ⭐ 67,3K + ⭐ 67.3K
Gemini CLI
- Gemini-CLI + Gemini CLI

- ⭐ 94,7K + ⭐ 94.7K
Kilo Code
- Kilo-Code + Kilo Code

- ⭐ 15,5K + ⭐ 15.5K
-📡 Alle Agenten verbinden sich über http://localhost:20128/v1 oder http://cloud.omniroute.online/v1 – eine Konfiguration, unbegrenzte Modelle und Kontingent--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Hören Sie auf, Geld zu verschwenden und an Grenzen zu stoßen:** +**Stop wasting money and hitting limits:** -- Das Abonnementkontingent läuft jeden Monat ungenutzt ab -- Ratenbeschränkungen stoppen Sie mitten beim Codieren - – Teure APIs (20–50 $/Monat pro Anbieter) -- Manueller Wechsel zwischen Anbietern +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute löst dieses Problem:** +**OmniRoute solves this:** -- ✅**Abonnements maximieren**- Verfolgen Sie das Kontingent, nutzen Sie jedes Bit vor dem Zurücksetzen -- ✅**Auto-Fallback**– Abonnement → API-Schlüssel → Günstig → Kostenlos, keine Ausfallzeiten -- ✅**Mehrere Konten**– Round-Robin zwischen Konten pro Anbieter -- ✅**Universell**– Funktioniert mit Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw und jedem CLI-Tool--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Treten Sie unserer Community bei!**[WhatsApp-Gruppe](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) – Holen Sie sich Hilfe, tauschen Sie Tipps aus und bleiben Sie auf dem Laufenden. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Website**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Probleme**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Community-Gruppe](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Mitwirken**: Siehe [CONTRIBUTING.md](CONTRIBUTING.md), öffnen Sie eine PR oder wählen Sie eine „gute erste Ausgabe“ aus -**Originalprojekt**: [9router von decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Wenn Sie ein Problem öffnen, führen Sie bitte den Befehl „system-info“ aus und hängen Sie die generierte Datei an:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Dadurch wird eine „system-info.txt“ mit Ihrer Node.js-Version, OmniRoute-Version, Betriebssystemdetails, installierten CLI-Tools (Qoder, Gemini, Claude, Codex, Antigravity, Droid usw.), Docker/PM2-Status und Systempaketen generiert – alles, was wir brauchen, um Ihr Problem schnell zu reproduzieren. Hängen Sie die Datei direkt an Ihr GitHub-Problem an.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Jeder Entwickler, der KI-Tools verwendet, ist täglich mit diesen Problemen konfrontiert.**OmniRoute wurde entwickelt, um sie alle zu lösen – von Kostenüberschreitungen bis hin zu regionalen Blockaden, von unterbrochenen OAuth-Flüssen bis hin zu Protokollvorgängen und Unternehmensbeobachtbarkeit. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. -
-💸 1. „Ich bezahle ein teures Abonnement, werde aber trotzdem durch Limits unterbrochen“ +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Entwickler zahlen 20–200 US-Dollar/Monat für Claude Pro, Codex Pro oder GitHub Copilot. Auch wenn das Kontingent bezahlt wird, gibt es eine Obergrenze – 5 Stunden Nutzung, wöchentliche Limits oder Tariflimits pro Minute. Während der Codierungssitzung reagiert der Anbieter nicht mehr und der Entwickler verliert an Fluss und Produktivität. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**So löst OmniRoute das Problem:** +**How OmniRoute solves it:** --**Intelligenter 4-Stufen-Fallback**– Wenn das Abonnementkontingent aufgebraucht ist, wird automatisch zu API Key → Günstig → Kostenlos weitergeleitet, ohne dass ein manueller Eingriff erforderlich ist --**Verfolgung von Anbieterlimits**– Zwischengespeicherte Kontingent-Snapshots werden nach einem serverseitigen Zeitplan aktualisiert (Standard „PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70“), wobei eine manuelle Aktualisierung in der Benutzeroberfläche verfügbar ist --**Unterstützung mehrerer Konten**– Mehrere Konten pro Anbieter mit automatischem Round-Robin – wenn eines aufgebraucht ist, wird zum nächsten gewechselt --**Benutzerdefinierte Kombinationen**– Anpassbare Fallback-Ketten mit 9 Ausgleichsstrategien (Priorität, gewichtet, Fill-First, Round-Robin, P2C, zufällig, am wenigsten genutzt, kostenoptimiert, strikt zufällig) --**Codex Business Quotas**– Überwachung der Geschäfts-/Team-Arbeitsbereichskontingente direkt im Dashboard
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard -
-🔌 2. „Ich muss mehrere Anbieter nutzen, aber jeder hat eine andere API“ +
-OpenAI verwendet ein Format, Claude (Anthropic) verwendet ein anderes, Gemini noch ein anderes. Wenn ein Entwickler Modelle verschiedener Anbieter testen oder zwischen ihnen wechseln möchte, muss er SDKs neu konfigurieren, Endpunkte ändern und mit inkompatiblen Formaten umgehen. Benutzerdefinierte Anbieter (FriendLI, NIM) verfügen über nicht standardmäßige Modellendpunkte. +
+🔌 2. "I need to use multiple providers but each has a different API" -**So löst OmniRoute das Problem:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Unified Endpoint**– Ein einzelner „http://localhost:20128/v1“ dient als Proxy für alle über 60 Anbieter --**Formatübersetzung**– Automatisch und transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API --**Antwortbereinigung**– Entfernt nicht standardmäßige Felder („x_groq“, „usage_breakdown“, „service_tier“), die OpenAI SDK v1.83+ beschädigen --**Rollennormalisierung**– Konvertiert „Entwickler“ → „System“ für Nicht-OpenAI-Anbieter; „System“ → „Benutzer“ für GLM/ERNIE --**Think Tag Extraction**– Extrahiert „“-Blöcke aus Modellen wie DeepSeek R1 in standardisierten „reasoning_content“. --**Strukturierte Ausgabe für Gemini**– automatische Konvertierung von „json_schema“ → „responseMimeType“/„responseSchema“. --**`stream` ist standardmäßig auf `false`**— Entspricht der OpenAI-Spezifikation und vermeidet unerwartetes SSE in Python/Rust/Go-SDKs
+**How OmniRoute solves it:** -
-🌐 3. „Mein KI-Anbieter blockiert meine Region/mein Land“ +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Anbieter wie OpenAI/Codex blockieren den Zugriff aus bestimmten geografischen Regionen. Benutzer erhalten bei OAuth- und API-Verbindungen Fehlermeldungen wie „unsupported_country_region_territory“. Dies ist besonders frustrierend für Entwickler aus Entwicklungsländern. +
-**So löst OmniRoute das Problem:** +
+🌐 3. "My AI provider blocks my region/country" --**3-Level-Proxy-Konfiguration**– Konfigurierbarer Proxy auf 3 Ebenen: global (gesamter Datenverkehr), pro Anbieter (nur ein Anbieter) und pro Verbindung/Schlüssel --**Farbcodierte Proxy-Abzeichen**– Visuelle Indikatoren: 🟢 globaler Proxy, 🟡 Anbieter-Proxy, 🔵 Verbindungs-Proxy, immer mit IP-Adresse --**OAuth-Token-Austausch über Proxy**– Der OAuth-Fluss läuft auch über den Proxy und löst „unsupported_country_region_territory“. --**Verbindungstests über Proxy**– Verbindungstests verwenden den konfigurierten Proxy (keine direkte Umgehung mehr) --**SOCKS5-Unterstützung**– Vollständige SOCKS5-Proxy-Unterstützung für ausgehendes Routing --**TLS-Fingerabdruck-Spoofing**– Browserähnlicher TLS-Fingerabdruck über „wreq-js“, um die Bot-Erkennung zu umgehen --**🔏 CLI-Fingerabdruck-Abgleich**– Ordnet Header und Textfelder neu an, damit sie mit nativen CLI-Binärsignaturen übereinstimmen, wodurch das Risiko der Kontokennzeichnung drastisch reduziert wird. Die Proxy-IP bleibt erhalten – Sie erhalten gleichzeitig Stealth**und**IP-Maskierung
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. -
-🆓 4. „Ich möchte KI zum Codieren verwenden, habe aber kein Geld“ +**How OmniRoute solves it:** -Nicht jeder kann 20–200 $/Monat für KI-Abonnements bezahlen. Studenten, Entwickler aus Schwellenländern, Bastler und Freiberufler benötigen Zugang zu hochwertigen Modellen zum Nulltarif. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**So löst OmniRoute das Problem:** +
--**Integrierte Free-Tier-Anbieter**– Native Unterstützung für 100 % kostenlose Anbieter: Qoder (5 unbegrenzte Modelle über OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unbegrenzte Modelle: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID kostenlos), Gemini CLI (180.000 Token/Monat kostenlos) --**Ollama Cloud**– Cloud-gehostete Ollama-Modelle unter „api.ollama.com“ mit kostenloser Stufe „Light-Nutzung“; Verwenden Sie das Präfix „ollamacloud/“. --**Nur kostenlose Combos**– Kette „gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus“ = 0 $/Monat ohne Ausfallzeit --**NVIDIA NIM Free Access**– Entwickler-für immer kostenloser Zugriff auf über 70 Modelle unter build.nvidia.com mit ca. 40 U/min (Umstellung von Credits auf reine Ratenlimits) --**Kostenoptimierte Strategie**– Routing-Strategie, die automatisch den günstigsten verfügbaren Anbieter auswählt
+
+🆓 4. "I want to use AI for coding but I have no money" -
-🔒 5. „Ich muss mein KI-Gateway vor unbefugtem Zugriff schützen“ +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -Wenn ein KI-Gateway dem Netzwerk (LAN, VPS, Docker) zugänglich gemacht wird, kann jeder mit der Adresse die Token/Kontingente des Entwicklers verbrauchen. Ohne Schutz sind APIs anfällig für Missbrauch, sofortige Injektion und Missbrauch. +**How OmniRoute solves it:** -**So löst OmniRoute das Problem:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**API-Schlüsselverwaltung**– Generierung, Rotation und Scoping pro Anbieter mit einer dedizierten „/dashboard/api-manager“-Seite --**Berechtigungen auf Modellebene**– Beschränken Sie API-Schlüssel auf bestimmte Modelle („openai/*“, Platzhaltermuster) mit der Umschaltfunktion „Alle zulassen/Einschränken“. --**API Endpoint Protection**– Erfordert einen Schlüssel für „/v1/models“ und blockiert bestimmte Anbieter aus der Liste --**Auth Guard + CSRF-Schutz**– Alle Dashboard-Routen sind mit „withAuth“-Middleware + CSRF-Tokens geschützt --**Ratenbegrenzer**– Ratenbegrenzung pro IP mit konfigurierbaren Fenstern --**IP-Filterung**– Zulassungs-/Blockierungsliste für die Zugriffskontrolle --**Prompt Injection Guard**– Bereinigung gegen bösartige Eingabeaufforderungsmuster --**AES-256-GCM-Verschlüsselung**– Anmeldeinformationen im Ruhezustand verschlüsselt
+
-
-🛑 6. „Mein Provider ist ausgefallen und ich habe meinen Programmierfluss verloren“ +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -KI-Anbieter können instabil werden, 5xx-Fehler zurückgeben oder vorübergehende Ratengrenzen erreichen. Wenn ein Entwickler von einem einzelnen Anbieter abhängig ist, wird er unterbrochen. Ohne Schutzschalter können wiederholte Versuche zum Absturz der Anwendung führen. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**So löst OmniRoute das Problem:** +**How OmniRoute solves it:** --**Leistungsschalter pro Modell**– Automatisches Öffnen/Schließen mit konfigurierbaren Schwellenwerten und Abklingzeit (Geschlossen/Offen/Halboffen), je nach Modell, um kaskadierende Blöcke zu vermeiden --**Exponentielles Backoff**– Progressive Wiederholungsverzögerungen --**Anti-Thundering Herd**– Mutex + Semaphor-Schutz gegen gleichzeitige Wiederholungsstürme --**Combo-Fallback-Ketten**– Wenn der primäre Anbieter ausfällt, fällt er automatisch durch die Kette, ohne dass ein Eingreifen erforderlich ist --**Combo Circuit Breaker**– Deaktiviert automatisch ausgefallene Anbieter innerhalb einer Combo-Kette --**Gesundheits-Dashboard**– Betriebszeitüberwachung, Leistungsschalterzustände, Sperren, Cache-Statistiken, p50/p95/p99-Latenz
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest -
-🔧 7. „Die Konfiguration jedes KI-Tools ist mühsam und repetitiv“ +
-Entwickler verwenden Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code ... Jedes Tool benötigt eine andere Konfiguration (API-Endpunkt, Schlüssel, Modell). Eine Neukonfiguration bei einem Anbieter- oder Modellwechsel ist Zeitverschwendung. +
+🛑 6. "My provider went down and I lost my coding flow" -**So löst OmniRoute das Problem:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**CLI Tools Dashboard**– Spezielle Seite mit Ein-Klick-Einrichtung für Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline --**GitHub Copilot Config Generator**– Erzeugt „chatLanguageModels.json“ für VS-Code mit Massenmodellauswahl --**Onboarding-Assistent**– Geführte Einrichtung in 4 Schritten für Erstbenutzer --**Ein Endpunkt, alle Modelle**– Konfigurieren Sie „http://localhost:20128/v1“ einmal und greifen Sie auf über 60 Anbieter zu
+**How OmniRoute solves it:** -
-🔑 8. „Die Verwaltung von OAuth-Tokens von mehreren Anbietern ist die Hölle“ +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot – alle verwenden OAuth 2.0 mit ablaufenden Token. Entwickler müssen sich ständig neu authentifizieren, sich mit „client_secret fehlt“, „redirect_uri_mismatch“ und Fehlern auf Remote-Servern befassen. Besonders problematisch ist OAuth auf LAN/VPS. +
-**So löst OmniRoute das Problem:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Automatische Token-Aktualisierung**– OAuth-Tokens werden vor Ablauf im Hintergrund aktualisiert --**OAuth 2.0 (PKCE) integriert**– Automatischer Ablauf für Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder --**Multi-Account OAuth**– Mehrere Konten pro Anbieter über JWT/ID-Token-Extraktion --**OAuth LAN/Remote Fix**– Private IP-Erkennung für „redirect_uri“ + manueller URL-Modus für Remote-Server --**OAuth hinter Nginx**– Verwendet „window.location.origin“ für Reverse-Proxy-Kompatibilität --**Remote OAuth Guide**– Schritt-für-Schritt-Anleitung für Google Cloud-Anmeldeinformationen auf VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. -
-📊 9. „Ich weiß nicht, wie viel ich wo ausgebe“ +**How OmniRoute solves it:** -Entwickler nutzen mehrere kostenpflichtige Anbieter, haben jedoch keine einheitliche Sicht auf die Ausgaben. Jeder Anbieter verfügt über ein eigenes Abrechnungs-Dashboard, es gibt jedoch keine konsolidierte Ansicht. Unerwartete Kosten können sich häufen. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**So löst OmniRoute das Problem:** +
--**Kostenanalyse-Dashboard**– Kostenverfolgung pro Token und Budgetverwaltung pro Anbieter --**Budgetgrenzen pro Stufe**– Ausgabenobergrenze pro Stufe, die einen automatischen Fallback auslöst --**Preiskonfiguration pro Modell**– Konfigurierbare Preise pro Modell --**Nutzungsstatistiken pro API-Schlüssel**– Anzahl der Anfragen und zuletzt verwendeter Zeitstempel pro Schlüssel --**Analytics-Dashboard**– Statistikkarten, Modellnutzungsdiagramm, Anbietertabelle mit Erfolgsraten und Latenz
+
+🔑 8. "Managing OAuth tokens from multiple providers is hell" -
-🐛 10. „Ich kann Fehler und Probleme bei KI-Aufrufen nicht diagnostizieren“ +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Wenn ein Anruf fehlschlägt, weiß der Entwickler nicht, ob es sich um eine Ratenbegrenzung, ein abgelaufenes Token, ein falsches Format oder einen Anbieterfehler handelt. Fragmentierte Protokolle über verschiedene Terminals hinweg. Ohne Beobachtbarkeit ist das Debuggen ein Versuch und Irrtum. +**How OmniRoute solves it:** -**So löst OmniRoute das Problem:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Einheitliches Protokoll-Dashboard**– 4 Registerkarten: Anforderungsprotokolle, Proxy-Protokolle, Audit-Protokolle, Konsole --**Console Log Viewer**– Echtzeit-Viewer im Terminal-Stil mit farbcodierten Ebenen, automatischem Scrollen, Suche und Filter --**SQLite-Proxy-Protokolle**– Persistente Protokolle, die Serverneustarts überdauern --**Translator Playground**– 4 Debugging-Modi: Playground (Formatübersetzung), Chat Tester (Round-Trip), Test Bench (Batch), Live Monitor (Echtzeit) --**Telemetrie anfordern**– p50/p95/p99-Latenz + X-Request-Id-Ablaufverfolgung --**Dateibasierte Protokollierung mit Rotation**– App-Protokolle rotieren nach Größe, Aufbewahrungstagen und Archivanzahl; Anrufprotokollartefakte rotieren nach Aufbewahrungstagen und Dateianzahl --**Systeminfobericht**– „npm run system-info“ generiert „system-info.txt“ mit Ihrer vollständigen Umgebung (Knotenversion, OmniRoute-Version, Betriebssystem, CLI-Tools, Docker/PM2-Status). Hängen Sie es an, wenn Sie Probleme melden, um eine sofortige Einstufung zu ermöglichen.
+
-
-🏗️ 11. „Die Bereitstellung und Wartung des Gateways ist komplex“ +
+📊 9. "I don't know how much I'm spending or where" -Die Installation, Konfiguration und Wartung eines KI-Proxys in verschiedenen Umgebungen (lokal, VPS, Docker, Cloud) ist arbeitsintensiv. Probleme wie hartcodierte Pfade, „EACCES“ für Verzeichnisse, Portkonflikte und plattformübergreifende Builds sorgen für zusätzliche Reibung. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**So löst OmniRoute das Problem:** +**How OmniRoute solves it:** --**npm globale Installation**– „npm install -g omniroute && omniroute“ – fertig --**Docker Multi-Platform**– AMD64 + ARM64 nativ (Apple Silicon, AWS Graviton, Raspberry Pi) --**Docker Compose-Profile**– „base“ (keine CLI-Tools) und „cli“ (mit Claude Code, Codex, OpenClaw) --**Electron Desktop App**– Native App für Windows/macOS/Linux mit Taskleiste, Autostart, Offline-Modus --**Split-Port-Modus**– API und Dashboard auf separaten Ports für erweiterte Szenarien (Reverse-Proxy, Container-Netzwerk) --**Cloud Sync**– Konfigurieren Sie die geräteübergreifende Synchronisierung über Cloudflare Workers --**DB-Backups**– Automatische Sicherung, Wiederherstellung, Export und Import aller Einstellungen, mit „DISABLE_SQLITE_AUTO_BACKUP“ für extern verwaltete Backups
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency -
-🌍 12. „Die Benutzeroberfläche ist nur auf Englisch verfügbar und mein Team spricht kein Englisch“ +
-Teams in nicht englischsprachigen Ländern, insbesondere in Lateinamerika, Asien und Europa, haben Probleme mit rein englischsprachigen Benutzeroberflächen. Sprachbarrieren verringern die Akzeptanz und erhöhen die Zahl von Konfigurationsfehlern. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**So löst OmniRoute das Problem:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Dashboard i18n – 30 Sprachen**– Alle über 500 Tasten übersetzt, einschließlich Arabisch, Bulgarisch, Dänisch, Deutsch, Spanisch, Finnisch, Französisch, Hebräisch, Hindi, Ungarisch, Indonesisch, Italienisch, Japanisch, Koreanisch, Malaiisch, Niederländisch, Norwegisch, Polnisch, Portugiesisch (PT/BR), Rumänisch, Russisch, Slowakisch, Schwedisch, Thailändisch, Ukrainisch, Vietnamesisch, Chinesisch, Philippinisch, Englisch --**RTL-Unterstützung**– Rechts-nach-links-Unterstützung für Arabisch und Hebräisch --**Mehrsprachige READMEs**– 30 vollständige Dokumentationsübersetzungen --**Sprachauswahl**– Globussymbol in der Kopfzeile zum Umschalten in Echtzeit
+**How OmniRoute solves it:** -
-🔄 13. „Ich brauche mehr als nur Chat – ich brauche Einbettungen, Bilder, Audio“ +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -KI ist nicht nur der Abschluss eines Chats. Entwickler müssen Bilder generieren, Audio transkribieren, Einbettungen für RAG erstellen, Dokumente neu einordnen und Inhalte moderieren. Jede API hat einen anderen Endpunkt und ein anderes Format. +
-**So löst OmniRoute das Problem:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Embeddings**– „/v1/embeddings“ mit 6 Anbietern und 9+ Modellen --**Image Generation**– „/v1/images/generations“ mit 10 Anbietern und über 20 Modellen (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Text-zu-Video**– „/v1/videos/generations“ – ComfyUI (AnimateDiff, SVD) und SD WebUI --**Text-zu-Musik**– „/v1/music/generations“ – ComfyUI (Stable Audio Open, MusicGen) --**Audiotranskription**– „/v1/audio/transcriptions“ – Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Text-to-Speech**– „/v1/audio/speech“ – ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + bestehende Anbieter --**Moderationen**– „/v1/moderations“ – Überprüfung der Inhaltssicherheit --**Reranking**– „/v1/rerank“ – Neuranking der Dokumentrelevanz --**Responses API**– Vollständige „/v1/responses“-Unterstützung für Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. -
-🧪 14. „Ich habe keine Möglichkeit, die Qualität verschiedener Modelle zu testen und zu vergleichen“ +**How OmniRoute solves it:** -Entwickler möchten wissen, welches Modell für ihren Anwendungsfall am besten geeignet ist – Code, Übersetzung, Argumentation –, aber ein manueller Vergleich ist langsam. Es sind keine integrierten Evaluierungstools vorhanden. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**So löst OmniRoute das Problem:** +
--**LLM-Bewertungen**– Golden-Set-Test mit 10 vorinstallierten Fällen zu Begrüßungen, Mathematik, Geografie, Codegenerierung, JSON-Konformität, Übersetzung, Markdown und Sicherheitsverweigerung --**4 Match-Strategien**– „exact“, „contains“, „regex“, „custom“ (JS-Funktion) --**Translator Playground Test Bench**– Batch-Tests mit mehreren Eingaben und erwarteten Ausgaben, anbieterübergreifender Vergleich --**Chat-Tester**– Vollständiger Roundtrip mit visueller Antwortwiedergabe --**Live-Monitor**– Echtzeit-Stream aller Anfragen, die über den Proxy fließen
+
+🌍 12. "The interface is English-only and my team doesn't speak English" -
-📈 15. „Ich muss skalieren, ohne an Leistung einzubüßen“ +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -Wenn das Anfragevolumen wächst, verursachen dieselben Fragen ohne Zwischenspeicherung doppelte Kosten. Ohne Idempotenz verschwenden doppelte Anfragen die Verarbeitung. Die Tarifbegrenzungen pro Anbieter müssen eingehalten werden. +**How OmniRoute solves it:** -**So löst OmniRoute das Problem:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Semantischer Cache**– Zweistufiger Cache (Signatur + Semantik) reduziert Kosten und Latenz --**Request Idempotency**– 5-Sekunden-Deduplizierungsfenster für identische Anfragen --**Ratenbegrenzungserkennung**– Provider-RPM, minimale Lücke und maximale gleichzeitige Verfolgung --**Bearbeitbare Ratengrenzen**– Konfigurierbare Standardeinstellungen unter Einstellungen → Ausfallsicherheit mit Persistenz --**API Key Validation Cache**– 3-stufiger Cache für Produktionsleistung --**Gesundheits-Dashboard mit Telemetrie**– p50/p95/p99-Latenz, Cache-Statistiken, Betriebszeit
+
-
-🤖 16. „Ich möchte das Modellverhalten global steuern“ +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Entwickler, die alle Antworten in einer bestimmten Sprache oder mit einem bestimmten Ton wünschen oder die Argumentationstoken einschränken möchten. Dies in jedem Tool/jeder Anfrage zu konfigurieren, ist unpraktisch. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**So löst OmniRoute das Problem:** +**How OmniRoute solves it:** --**System Prompt Injection**– Globale Eingabeaufforderung, die auf alle Anfragen angewendet wird --**Thinking Budget Validation**– Reasoning-Token-Zuteilungskontrolle pro Anfrage (Passthrough, automatisch, benutzerdefiniert, adaptiv) --**9 Routing-Strategien**– Globale Strategien, die bestimmen, wie Anfragen verteilt werden --**Wildcard-Router**– „provider/*“-Muster leiten dynamisch an jeden Anbieter weiter --**Combo-Aktivierung/Deaktivierung umschalten**– Combos direkt über das Dashboard umschalten --**Provider Toggle**– Alle Verbindungen für einen Anbieter mit einem Klick aktivieren/deaktivieren --**Blockierte Anbieter**– Bestimmte Anbieter aus der Liste „/v1/models“ ausschließen
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex -
-🧰 17. „Ich brauche MCP-Tools als erstklassige Produktfunktionen“ +
-Viele KI-Gateways stellen MCP nur als verstecktes Implementierungsdetail zur Verfügung. Teams benötigen eine sichtbare, überschaubare Betriebsebene. +
+🧪 14. "I have no way to test and compare quality across models" -**So löst OmniRoute das Problem:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -– MCP wird in der Dashboard-Navigation und auf der Registerkarte „Endpunktprotokoll“ angezeigt -- Dedizierte MCP-Verwaltungsseite mit Prozess, Tools, Bereichen und Audit -– Integrierter Schnellstart für „omniroute --mcp“ und Client-Onboarding
+**How OmniRoute solves it:** -
-🧠 18. „Ich benötige A2A-Orchestrierung mit Synchronisierungs- und Stream-Aufgabenpfaden“ +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Agenten-Workflows erfordern sowohl direkte Antworten als auch eine lang andauernde gestreamte Ausführung mit Lebenszykluskontrolle. +
-**So löst OmniRoute das Problem:** +
+📈 15. "I need to scale without losing performance" -- A2A JSON-RPC-Endpunkt („POST /a2a“) mit „message/send“ und „message/stream“. -- SSE-Streaming mit Terminal-State-Propagierung -– Task-Lebenszyklus-APIs für „tasks/get“ und „tasks/cancel“.
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. -
-🛰️ 19. „Ich brauche einen echten Zustand des MCP-Prozesses, keinen erratenen Status“ +**How OmniRoute solves it:** -Betriebsteams müssen wissen, ob MCP tatsächlich aktiv ist, und nicht nur, ob eine API erreichbar ist. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**So löst OmniRoute das Problem:** +
-– Laufzeit-Heartbeat-Datei mit PID, Zeitstempeln, Transport, Werkzeuganzahl und Oszilloskopmodus -- MCP-Status-API, die Heartbeat + aktuelle Aktivität kombiniert -- UI-Statuskarten für Prozess-/Verfügbarkeits-/Heartbeat-Aktualität
+
+🤖 16. "I want to control model behavior globally" -
-📋 20. „Ich benötige eine überprüfbare MCP-Tool-Ausführung“ +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Wenn Tools die Konfiguration verändern oder operative Aktionen auslösen, benötigen Teams forensische Rückverfolgbarkeit. +**How OmniRoute solves it:** -**So löst OmniRoute das Problem:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -– SQLite-gestützte Audit-Protokollierung für MCP-Tool-Aufrufe -- Filtert nach Tool, Erfolg/Misserfolg, API-Schlüssel und Paginierung -- Dashboard-Audit-Tabelle + Statistik-Endpunkte für die Automatisierung
+
-
-🔐 21. „Ich benötige bereichsbezogene MCP-Berechtigungen pro Integration“ +
+🧰 17. "I need MCP tools as first-class product capabilities" -Verschiedene Clients sollten Zugriff auf die Werkzeugkategorien mit den geringsten Rechten haben. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**So löst OmniRoute das Problem:** +**How OmniRoute solves it:** -- 10 granulare MCP-Bereiche für kontrollierten Werkzeugzugriff -- Geltungsbereichsdurchsetzung und Sichtbarkeit in der MCP-Management-Benutzeroberfläche -- Sichere Standardhaltung für Betriebswerkzeuge
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding -
-⚙️ 22. „Ich brauche Betriebskontrollen ohne Umschichtung“ +
-Teams benötigen bei Vorfällen oder Kostenereignissen schnelle Laufzeitänderungen. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**So löst OmniRoute das Problem:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Schalten Sie die Combo-Aktivierung direkt über das MCP-Dashboard um -- Wenden Sie Ausfallsicherheitsprofile aus vordefinierten Richtlinienpaketen an -- Setzen Sie den Leistungsschalterstatus über dasselbe Bedienfeld zurück
+**How OmniRoute solves it:** -
-🔄 23. „Ich benötige Live-Sichtbarkeit und Stornierung des A2A-Aufgabenlebenszyklus“ +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Ohne Sichtbarkeit des Lebenszyklus wird es schwierig, Aufgabenvorfälle zu selektieren. +
-**So löst OmniRoute das Problem:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Aufgabenliste/Filterung nach Bundesland/Fähigkeit mit Paginierung -- Drilldown zu Aufgabenmetadaten, Ereignissen und Artefakten -- Endpunkt zum Abbrechen von Aufgaben und UI-Aktion mit Bestätigung
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. -
-🌊 24. „Ich benötige aktive Stream-Metriken für die A2A-Last“ +**How OmniRoute solves it:** -Streaming-Workflows erfordern betriebliche Einblicke in Parallelität und Live-Verbindungen. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**So löst OmniRoute das Problem:** +
-- Aktive Stream-Zähler im A2A-Status integriert -- Zeitstempel der letzten Aufgabe und Anzahl pro Status -- A2A-Dashboard-Karten für die Echtzeit-Betriebsüberwachung
+
+📋 20. "I need auditable MCP tool execution" -
-🪪 25. „Ich benötige eine standardmäßige Agentenerkennung für Kunden“ +When tools mutate config or trigger ops actions, teams need forensic traceability. -Externe Kunden und Orchestratoren benötigen für das Onboarding maschinenlesbare Metadaten. +**How OmniRoute solves it:** -**So löst OmniRoute das Problem:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -– Agentenkarte unter „/.well-known/agent.json“ verfügbar gemacht -- Fähigkeiten und Fertigkeiten werden in der Management-Benutzeroberfläche angezeigt -– Die A2A-Status-API enthält Erkennungsmetadaten für die Automatisierung
+
-
-🧭 26. „Ich benötige Protokollauffindbarkeit in der Produkt-UX“ +
+🔐 21. "I need scoped MCP permissions per integration" -Wenn Benutzer Protokolloberflächen nicht entdecken können, sinken Akzeptanz und Supportqualität. +Different clients should have least-privilege access to tool categories. -**So löst OmniRoute das Problem:** +**How OmniRoute solves it:** -- Konsolidierte Seite**Endpunkte**mit Registerkarten für Proxy-, MCP-, A2A- und API-Endpunkte -- Inline-Dienststatusumschaltung (Online/Offline) für MCP und A2A -- Links von der Übersicht zu speziellen Verwaltungsregisterkarten
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling -
-🧪 27. „Ich benötige eine End-to-End-Protokollvalidierung mit echten Clients“ +
-Probetests reichen nicht aus, um die Protokollkompatibilität vor der Veröffentlichung zu überprüfen. +
+⚙️ 22. "I need operational controls without redeploying" -**So löst OmniRoute das Problem:** +Teams need quick runtime changes during incidents or cost events. -– E2E-Suite, die die App startet und echten MCP SDK-Client-Transport verwendet -- A2A-Clienttests für Erkennungs-, Sende-, Stream-, Get- und Abbruchflüsse -- Vergleichen Sie Behauptungen mit MCP-Audit- und A2A-Aufgaben-APIs
+**How OmniRoute solves it:** -
-📡 28. „Ich brauche eine einheitliche Beobachtbarkeit über alle Schnittstellen hinweg“ +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -Die Aufteilung der Beobachtbarkeit nach Protokoll führt zu blinden Flecken und einer längeren MTTR. +
-**So löst OmniRoute das Problem:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Einheitliche Dashboards/Protokolle/Analysen in einem Produkt -- Gesundheits-, Audit- und Anforderungstelemetrie über OpenAI-, MCP- und A2A-Ebenen hinweg -- Operative APIs für Status und Automatisierung
+Without lifecycle visibility, task incidents become hard to triage. -
-💼 29. „Ich benötige eine Laufzeit für Proxy + Tools + Agent-Orchestrierung“ +**How OmniRoute solves it:** -Die Ausführung vieler separater Dienste erhöht die Betriebskosten und erhöht die Fehlerhäufigkeit. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**So löst OmniRoute das Problem:** +
-- OpenAI-kompatibler Proxy, MCP-Server und A2A-Server in einem Stack -– Gemeinsame Authentifizierung, Ausfallsicherheit, Datenspeicher und Beobachtbarkeit -- Konsistentes Richtlinienmodell über alle Interaktionsoberflächen hinweg
+
+🌊 24. "I need active stream metrics for A2A load" -
-🚀 30. „Ich muss Agenten-Workflows ohne Glue-Code-Wildwuchs ausliefern“ +Streaming workflows require operational insight into concurrency and live connections. -Teams verlieren an Geschwindigkeit, wenn sie mehrere Ad-hoc-Dienste und -Skripte zusammenfügen. +**How OmniRoute solves it:** -**So löst OmniRoute das Problem:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Einheitliche Endpunktstrategie für Kunden und Agenten -- Integrierte Protokollverwaltungs-Benutzeroberflächen und Rauchvalidierungspfade -- Produktionsreife Grundlagen (Sicherheit, Protokollierung, Ausfallsicherheit, Backup)
+
+ +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Playbook A: Bezahltes Abonnement maximieren + günstiges Backup**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -607,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Playbook B: Kostenfreier Codierungsstack**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Playbook C: 24/7 Always-On-Fallback-Kette**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -630,125 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Playbook D: Agenteneinsätze mit MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Richten Sie die KI-Codierung in wenigen Minuten für**0 $/Monat**ein. Verbinden Sie diese kostenlosen Konten und nutzen Sie die integrierte**Free Stack**-Kombination. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Schritt | Aktion | Anbieter freigeschaltet | -| ---- | ------------------------------------------------- | ----------------------------------------------------------------- | -| 1 | Verbinden Sie**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 –**unbegrenzt**| -| 2 | Verbinden Sie**Qoder**(Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... —**unbegrenzt**| -| 3 | Verbinden Sie**Qwen**(Gerätecode) | qwen3-coder-plus, qwen3-coder-flash... —**unbegrenzt**| -| 4 | Verbinden Sie**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro –**180.000/Monat kostenlos**| -| 5 | `/dashboard/combos` → Vorlage**Free Stack ($0)**| Round-Robin aller kostenlosen Anbieter automatisch | +| Step | Action | Providers Unlocked | +| ---- | -------------------------------------------------- | ------------------------------------------------------------------ | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Zeigen Sie eine beliebige IDE/CLI auf:**„http://localhost:20128/v1“ · API-Schlüssel: „any-string“ · Fertig. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Optionale zusätzliche Abdeckung (auch kostenlos):**Groq API-Schlüssel (30 U/min kostenlos), NVIDIA NIM (40 U/min kostenlos, 70+ Modelle), Cerebras (1 Mio. Token/Tag), LongCat API-Schlüssel (50 Mio. Token/Tag!), Cloudflare Workers AI (10.000 Neuronen/Tag, 50+ Modelle).## Schnellstart +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Schnellstart ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **pnpm-Benutzer:**Führen Sie nach der Installation „pnpm genehmigt-builds -g“ aus, um native Build-Skripte zu aktivieren, die für „better-sqlite3“ und „@swc/core“ erforderlich sind: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > -> „Bash +> ```bash > pnpm install -g omniroute -> pnpm genehmigt-builds -g # Alle Pakete auswählen → genehmigen -> Omniroute -> -> ``` -> +> pnpm approve-builds -g # Select all packages → approve +> omniroute > ``` -Das Dashboard wird unter „http://localhost:20128“ geöffnet und die API-Basis-URL ist „http://localhost:20128/v1“. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Befehl | Beschreibung | -| ----------------------- | ------------------------------------------------------------------- | -| `omniroute` | Server starten („PORT=20128“, API und Dashboard auf demselben Port) | -| `omniroute --port 3000` | Setzen Sie den kanonischen/API-Port auf 3000 | -| `omniroute --mcp` | Starten Sie den MCP-Server (STDIO-Transport) | -| `omniroute --no-open` | Browser nicht automatisch öffnen | -| `omniroute --help` | Hilfe anzeigen | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Optionaler Split-Port-Modus:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -Für die meisten Bereitstellungen benötigen Sie lediglich: +For most deployments, you only need: -| Variable | Standard | Zweck | -| ------------------------ | -------------- | ---------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | „600000“ | Gemeinsame Baseline für Upstream-Abruf, versteckte Undici-Timeouts, TLS-Fingerprint-Anfragen und API-Bridge-Request/Proxy-Timeouts | -| `STREAM_IDLE_TIMEOUT_MS` | erbt „REQUEST_TIMEOUT_MS“ | Maximale Lücke zwischen Streaming-Blöcken, bevor OmniRoute den SSE-Stream abbricht | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -Die Abwärtskompatibilität bleibt erhalten: Vorhandene „FETCH_TIMEOUT_MS“, „API_BRIDGE_PROXY_TIMEOUT_MS“ und andere Timeout-Variablen pro Ebene funktionieren weiterhin und überschreiben die gemeinsame Baseline. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Wenn Sie eine genauere Steuerung benötigen, stehen erweiterte Überschreibungen zur Verfügung:| Variable | Standard | Zweck | -| ---------------------------------------- | ------------------------------------------ | ------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | erbt „REQUEST_TIMEOUT_MS“ | Gesamtzeitüberschreitung der Upstream-Anforderung, die vom Hauptabrufsignal | verwendet wird -| `FETCH_HEADERS_TIMEOUT_MS` | erbt „FETCH_TIMEOUT_MS“ | Undici-Zeitlimit für den Empfang von Upstream-Antwortheadern | -| `FETCH_BODY_TIMEOUT_MS` | erbt „FETCH_TIMEOUT_MS“ | Undici-Zeitlimit zwischen Upstream-Body-Chunks („0“ deaktiviert es) | -| `FETCH_CONNECT_TIMEOUT_MS` | „30000“ | Undici TCP-Verbindungszeitüberschreitung | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | „4000“ | Undici Leerlauf-Keep-Alive-Socket-Timeout | -| `TLS_CLIENT_TIMEOUT_MS` | erbt „FETCH_TIMEOUT_MS“ | Zeitüberschreitung für TLS-Fingerabdruckanfragen über „wreq-js“ | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | erbt „REQUEST_TIMEOUT_MS“ oder „30000“ | Zeitüberschreitung für „/v1“-Proxy-Weiterleitung vom API-Port zum Dashboard-Port | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Zeitüberschreitung bei eingehenden Anfragen auf dem API-Bridge-Server | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | „60000“ | Zeitüberschreitung beim eingehenden Header auf dem API-Bridge-Server | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | „5000“ | Keep-Alive-Timeout auf dem API-Bridge-Server | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Zeitüberschreitung bei Socket-Inaktivität auf dem API-Bridge-Server („0“ deaktiviert ihn) | +Advanced overrides are available if you need finer control: -Wenn Sie OmniRoute hinter Nginx, Caddy, Cloudflare oder einem anderen Reverse-Proxy ausführen, stellen Sie sicher, dass der Proxy vorhanden ist -Die Zeitüberschreitungen sind auch höher als die Zeitüberschreitungen für Ihren OmniRoute-Stream/Abruf.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Öffnen Sie Dashboard → „Anbieter“ und verbinden Sie mindestens einen Anbieter (OAuth oder API-Schlüssel). -2. Öffnen Sie Dashboard → „Endpunkte“ und erstellen Sie einen API-Schlüssel. -3. (Optional) Öffnen Sie Dashboard → „Combos“ und legen Sie Ihre Fallback-Kette fest.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Funktioniert mit Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode und OpenAI-kompatiblen SDKs.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (für werkzeuggesteuerte Vorgänge):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -Verbinden Sie dann Ihren MCP-Client über „stdio“ und testen Sie Tools wie: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (für Agent-zu-Agent-Workflows):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -762,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Diese Suite validiert echte MCP- und A2A-Client-Flows anhand einer laufenden App.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -770,13 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` -
-Void Linux (Vorlage „xbps-src“) +
+Void Linux (`xbps-src` template) -Für Void-Linux-Benutzer können Sie mit „xbps-src“ ein natives Paket erstellen. Speichern Sie diesen Block als „srcpkgs/omniroute/template“:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -788,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -796,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -872,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -883,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute ist als öffentliches Docker-Image auf [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute) verfügbar. +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Schneller Lauf:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -893,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**Mit Umgebungsdatei:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Verwendung von Docker Compose:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Die Dashboard-Unterstützung für Docker-Bereitstellungen umfasst jetzt einen**Cloudflare Quick Tunnel**mit einem Klick unter „Dashboard → Endpunkte“. Die erste Aktivierung lädt „cloudflared“ nur bei Bedarf herunter, startet einen temporären Tunnel zu Ihrem aktuellen „/v1“-Endpunkt und zeigt die generierte „https://\*.trycloudflare.com/v1“-URL direkt unter Ihrer normalen öffentlichen URL an. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Hinweise: +Notes: -- Quick Tunnel-URLs sind temporär und ändern sich nach jedem Neustart. - – Quick Tunnels werden nach einem OmniRoute- oder Container-Neustart nicht automatisch wiederhergestellt. Aktivieren Sie sie bei Bedarf über das Dashboard erneut. - – Die verwaltete Installation unterstützt derzeit Linux, macOS und Windows auf „x64“ / „arm64“. - – Managed Quick Tunnels verwenden standardmäßig den HTTP/2-Transport, um laute QUIC-UDP-Pufferwarnungen in eingeschränkten Containerumgebungen zu vermeiden. Stellen Sie „CLOUDFLARED_PROTOCOL=quic“ oder „auto“ ein, wenn Sie einen anderen Transport wünschen. -- Docker-Images bündeln System-CA-Roots und übergeben sie an verwaltetes „Cloudflared“, wodurch TLS-Vertrauensfehler vermieden werden, wenn der Tunnel innerhalb des Containers bootet. -- SQLite läuft im WAL-Modus. „Docker Stop“ sollte abgeschlossen werden dürfen, damit OmniRoute die neuesten Änderungen zurück in „storage.sqlite“ überprüfen kann. - – Die gebündelten Compose-Dateien legen bereits eine Stoppfrist von 40 Sekunden fest. Wenn Sie das Image direkt ausführen, behalten Sie „--stop-timeout 40“ (oder ähnlich) bei, damit manuelle Stopps die Bereinigung beim Herunterfahren nicht unterbrechen. -- Legen Sie „CLOUDFLARED_BIN=/absolute/path/to/cloudflared“ fest, wenn OmniRoute eine vorhandene Binärdatei verwenden soll, anstatt eine herunterzuladen. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Verwendung von Docker Compose mit Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute kann mithilfe der automatischen SSL-Bereitstellung von Caddy sicher verfügbar gemacht werden. Stellen Sie sicher, dass der DNS-A-Eintrag Ihrer Domain auf die IP Ihres Servers verweist.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` - -| Bild | Tag | Größe | Beschreibung | +| Image | Tag | Size | Description | | ------------------------ | -------- | ------ | --------------------- | -| `diegosouzapw/omniroute` | `neueste` | ~250 MB | Neueste stabile Version | -| `diegosouzapw/omniroute` | `1.0.3` | ~250 MB | Aktuelle Version |--- +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | + +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**NEU!**OmniRoute ist jetzt als**native Desktop-Anwendung**für Windows, macOS und Linux verfügbar. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Führen Sie OmniRoute als eigenständige Desktop-App aus – kein Terminal, kein Browser, keine Internetverbindung für lokale Modelle erforderlich. Die Electron-basierte App umfasst: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Natives Fenster**– Spezielles App-Fenster mit Integration in die Taskleiste -- 🔄**Auto-Start**– OmniRoute bei der Systemanmeldung starten -- 🔔**Native Benachrichtigungen**– Erhalten Sie Benachrichtigungen bei Kontingentausschöpfung oder Anbieterproblemen -- ⚡**One-Click-Installation**– NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Offline-Modus**– Funktioniert vollständig offline mit dem gebündelten Server### Schnellstart +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Schnellstart ```bash # Development mode @@ -982,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -Wenn OmniRoute minimiert ist, befindet es sich mit schnellen Aktionen in Ihrer Taskleiste: +When minimized, OmniRoute lives in your system tray with quick actions: -- Dashboard öffnen -- Server-Port ändern -- Anwendung beenden +- Open dashboard +- Change server port +- Quit application -📖 Vollständige Dokumentation: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Stufe | Anbieter | Kosten | Kontingent zurücksetzen | Am besten für | -| -------------------- | --------------------------- | ---------------------------------------- | ------------------------- | ------------------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 ABO** | Claude Code (Pro) | 20 $/Monat | 5h + wöchentlich | Bereits abonniert | -| | Codex (Plus/Pro) | 20–200 $/Monat | 5h + wöchentlich | OpenAI-Benutzer | -| | Gemini CLI | **KOSTENLOS** | 180.000/Monat + 1.000/Tag | Alle! | -| | GitHub-Copilot | 10–19 $/Monat | Monatlich | GitHub-Benutzer | -| **🔑 API-SCHLÜSSEL** | NVIDIA NIM | **KOSTENLOS**(für immer entwickeln) | ~40 U/min | Über 70 offene Modelle | -| | Großhirn | **KOSTENLOS**(1 Mio. tok/Tag) | 60.000 TPM / 30 U/min | Der schnellste der Welt | -| | Groq | **KOSTENLOS**(30 U/min) | 14,4K RPD | Ultraschnelles Lama/Gemma | -| | DeepSeek V3.2 | 0,27 $/1,10 $ pro 1 Mio. | Keine | Bestes Preis-Leistungs-Verhältnis | -| | xAI Grok-4 Schnell | **0,20 $/0,50 $ pro 1 Mio.**🆕 | Keine | Schnellster + Werkzeugaufruf, ultraniedrig | -| | xAI Grok-4 (Standard) | 0,20 $/1,50 $ pro 1 Mio. 🆕 | Keine | Argumentations-Flaggschiff von xAI | -| | Mistral | Kostenlose Testversion + kostenpflichtig | Tarif begrenzt | Europäische KI | -| | OpenRouter | Pay-per-Use | Keine | Über 100 Modelle aggr. | -| **💰 GÜNSTIG** | GLM-5 (über Z.AI) 🆕 | 0,5 $/1 Mio. | Täglich 10 Uhr | 128K-Ausgabe, neuestes Flaggschiff | -| | GLM-4.7 | 0,6 $/1 Mio. | Täglich 10 Uhr | Budgetsicherung | -| | MiniMax M2.5 🆕 | 0,3 $/1 Mio. Eingabe | 5-Stunden-Rollen | Argumentation + Agentenaufgaben | -| | MiniMax M2.1 | 0,2 $/1 Mio. | 5-Stunden-Rollen | Günstigste Option | -| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-Use | Keine | Direkter Zugriff auf die Moonshot-API | -| | Kimi K2 | $9/Monat pauschal | 10 Millionen Token/Monat | Vorhersehbare Kosten | -| **🆓 KOSTENLOS** | Qoder | **$0** | Unbegrenzt | 5 Modelle unbegrenzt | -| | Qwen | **$0** | Unbegrenzt | 4 Modelle unbegrenzt | -| | Kiro | **$0** | Unbegrenzt | Claude Sonnet/Haiku (AWS Builder) | -| | LongCat Flash-Lite 🆕 | **$0**(50 Mio. Token/Tag 🔥) | 1 RPS | Größte kostenlose Quote der Welt | -| | Bestäubungs-KI 🆕 | **$0**(kein Schlüssel erforderlich) | 1 Anforderung/15s | GPT-5, Claude, DeepSeek, Lama 4 | -| | Cloudflare Workers AI 🆕 | **$0**(10.000 Neuronen/Tag) | ~150 resp/Tag | Über 50 Modelle, globaler Vorsprung | -| | Scaleway AI 🆕 | **0 $**(insgesamt 1 Mio. Token) | Tarif begrenzt | EU/DSGVO, Qwen3 235B, Lama 70B | > 🆕**Neue Modelle hinzugefügt (März 2026):**Grok-4 Fast-Familie für 0,20 $/0,50 $/M (Benchmark bei 1143 ms – 30 % schneller als Gemini 2.5 Flash), GLM-5 über Z.AI mit 128K-Ausgabe, MiniMax M2.5-Argumentation, aktualisierte Preise für DeepSeek V3.2, Kimi K2.5 über die direkte Moonshot-API. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 0 $ Combo Stack – Das komplette kostenlose Setup:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Kostenlos. Hört nie auf zu programmieren.**Konfigurieren Sie dies als eine OmniRoute-Kombination und alle Fallbacks erfolgen automatisch – kein manuelles Umschalten.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Alle unten aufgeführten Modelle sind**100 % kostenlos, keine Kreditkarte erforderlich**. OmniRoute leitet automatisch zwischen ihnen weiter, wenn ein Kontingent aufgebraucht ist – kombinieren Sie sie alle für eine unzerstörbare 0-Dollar-Kombination.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Modell | Präfix | Grenze | Ratenbegrenzung | +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | --------------------- | -| `claude-sonett-4.5` | `kr/` |**Unbegrenzt**| Keine gemeldete Tagesobergrenze | -| `claude-haiku-4.5` | `kr/` |**Unbegrenzt**| Keine gemeldete Tagesobergrenze | -| `claude-opus-4.6` | `kr/` |**Unbegrenzt**| Neuestes Werk von Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -| Modell | Präfix | Grenze | Ratenbegrenzung | -| ------------------- | ------ | ------------- | --------------- | -| `kimi-k2-thinking` | `if/` |**Unbegrenzt**| Keine gemeldete Obergrenze | -| `qwen3-coder-plus` | `if/` |**Unbegrenzt**| Keine gemeldete Obergrenze | -| `deepseek-r1` | `if/` |**Unbegrenzt**| Keine gemeldete Obergrenze | -| `minimax-m2.1` | `if/` |**Unbegrenzt**| Keine gemeldete Obergrenze | -| `kimi-k2` | `if/` |**Unbegrenzt**| Keine gemeldete Obergrenze | +### 🟢 QODER MODELS (Free PAT via qodercli) -> Empfohlene Verbindungsmethode:**Persönliches Zugriffstoken + „qodercli“**. Browser OAuth ist -> experimentell und standardmäßig deaktiviert, es sei denn, die Umgebungsvariablen „QODER_OAUTH_*“ sind konfiguriert.### 🟡 QWEN MODELS (Device Code Auth) +| Model | Prefix | Limit | Rate Limit | +| ------------------ | ------ | ------------- | --------------- | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -| Modell | Präfix | Grenze | Ratenlimit | +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. + +### 🟡 QWEN MODELS (Device Code Auth) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | ------------------- | -| `qwen3-coder-plus` | `qw/` |**Unbegrenzt**| Keine gemeldete Obergrenze | -| `qwen3-coder-flash` | `qw/` |**Unbegrenzt**| Keine gemeldete Obergrenze | -| `qwen3-coder-next` | `qw/` |**Unbegrenzt**| Keine gemeldete Obergrenze | -| „Vision-Modell“ | `qw/` |**Unbegrenzt**| Multimodal (Bilder) |### 🟣 GEMINI CLI (Google OAuth) +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Modell | Präfix | Grenze | Ratenlimit | -| ------------------------ | ------ | ------------ | ------------- | -| `gemini-3-flash-preview` | `gc/` |**180.000 Token/Monat**+ 1.000/Tag | Monatlicher Reset | -| `gemini-2.5-pro` | `gc/` | 180.000/Monat (gemeinsamer Pool) | Hohe Qualität |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +### 🟣 GEMINI CLI (Google OAuth) -| Stufe | Tageslimit | Ratenlimit | Notizen | -| ---------- | ------------ | ----------- | ----------------------------------------------------- | -| Kostenlos (Entwickler) | Keine Token-Obergrenze |**~40 U/min**| Über 70 Modelle; Übergang zu reinen Tarifbegrenzungen Mitte 2025 | +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -Beliebte kostenlose Modelle: „moonshotai/kimi-k2.5“ (Kimi K2.5), „z-ai/glm4.7“ (GLM 4.7), „deepseek-ai/deepseek-v3.2“ (DeepSeek V3.2), „nvidia/llama-3.3-70b-instruct“, „deepseek/deepseek-r1“.### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) -| Stufe | Tageslimit | Ratenlimit | Notizen | +| Tier | Daily Limit | Rate Limit | Notes | +| ---------- | ------------ | ----------- | ------------------------------------------------------ | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | + +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` + +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) + +| Tier | Daily Limit | Rate Limit | Notes | | ---- | ----------------- | ---------------- | ------------------------------------------- | -| Kostenlos |**1 Mio. Token/Tag**| 60.000 TPM / 30 U/min | Weltweit schnellste LLM-Inferenz; wird täglich zurückgesetzt | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -Kostenlos erhältlich: „llama-3.3-70b“, „llama-3.1-8b“, „deepseek-r1-distill-llama-70b“.### 🔴 GROQ (Free API Key — console.groq.com) +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -| Stufe | Tageslimit | Ratenlimit | Notizen | +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---- | ------------- | ---------------- | ----------------------------------------- | -| Kostenlos |**14,4K RPD**| 30 U/min pro Modell | Keine Kreditkarte; 429 auf Limit, nicht berechnet | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -Kostenlos erhältlich: „llama-3.3-70b-versatile“, „gemma2-9b-it“, „mixtral-8x7b“, „whisper-large-v3“.### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -| Modell | Präfix | Tägliches kostenloses Kontingent | Notizen | -| -------------- | ------ | ----------------- | --------- | -| `LongCat-Flash-Lite` | `lc/` |**50 Millionen Token**💥 | Größtes kostenloses Kontingent aller Zeiten | -| `LongCat-Flash-Chat` | `lc/` | 500.000 Token | Multi-Turn-Chat | -| „LongCat-Flash-Thinking“ | `lc/` | 500.000 Token | Begründung / CoT | -| `LongCat-Flash-Thinking-2601` | `lc/` | 500.000 Token | Version Januar 2026 | -| „LongCat-Flash-Omni-2603“ | `lc/` | 500.000 Token | Multimodal | +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 -> 100 % kostenlos während der öffentlichen Beta. Melden Sie sich per E-Mail oder Telefon bei [longcat.chat](https://longcat.chat) an. Wird täglich um 00:00 UTC zurückgesetzt.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | -| Modell | Präfix | Ratenlimit | Anbieter dahinter | -| ---------- | ------ | ---------- | ------------------- | -| `openai` | `pol/` | 1 Anforderung/15s | GPT-5 | -| `Claude` | `pol/` | 1 Anforderung/15s | Anthropischer Claude | -| „Zwillinge“ | `pol/` | 1 Anforderung/15s | Google Gemini | -| `deepseek` | `pol/` | 1 Anforderung/15s | DeepSeek V3 | -| `Lama` | `pol/` | 1 Anforderung/15s | Meta Lama 4 Scout | -| „Mistral“ | `pol/` | 1 Anforderung/15s | Mistral KI | +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. -> ✨**Keine Reibung:**Keine Anmeldung, kein API-Schlüssel. Fügen Sie den Bestäubungsanbieter mit einem leeren Schlüsselfeld hinzu und es funktioniert sofort.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 -| Stufe | Tägliche Neuronen | Äquivalente Verwendung | Notizen | -| ---- | ------------- | --------------------------------------- | --------- | -| Kostenlos |**10.000**| ~150 LLM bzw. 500 Sek. Audio / 15.000 Einbettungen | Global Edge, 50+ Modelle | +| Model | Prefix | Rate Limit | Provider Behind | +| ---------- | ------ | ---------- | ------------------ | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -Beliebte kostenlose Modelle: „@cf/meta/llama-3.3-70b-instruct“, „@cf/google/gemma-3-12b-it“, „@cf/openai/whisper-large-v3-turbo“ (kostenloses Audio!), „@cf/qwen/qwen2.5-coder-15b-instruct“. +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -> Erfordert API-Token + Konto-ID von [dash.cloudflare.com](https://dash.cloudflare.com). Konto-ID in den Anbietereinstellungen hinterlegen.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 -| Stufe | Kostenloses Kontingent | Standort | Notizen | +| Tier | Daily Neurons | Equivalent Usage | Notes | +| ---- | ------------- | --------------------------------------- | ----------------------- | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | + +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` + +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. + +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | | ---- | ------------- | ------------ | ----------------------------------- | -| Kostenlos |**1 Mio. Token**| 🇫🇷 Paris, EU | Innerhalb der Grenzen ist keine Kreditkarte erforderlich | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | -Kostenlos verfügbar: „qwen3-235b-a22b-instruct-2507“ (Qwen3 235B!), „llama-3.1-70b-instruct“, „mistral-small-3.2-24b-instruct-2506“, „deepseek-v3-0324“. +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` -> EU/DSGVO-konform. Holen Sie sich den API-Schlüssel unter [console.scaleway.com](https://console.scaleway.com). +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). ->**💡 Der ultimative kostenlose Stack (11 Anbieter, 0 $ für immer):** +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -> LongCat Lite (lc/) → LongCat-Flash-Lite – 50 Millionen Token/Tag 🔥 -> Bestäubungen (pol/) → GPT-5, Claude, DeepSeek, Llama 4 – kein Schlüssel erforderlich -> Qwen (qw/) → qwen3-Coder-Modelle UNBEGRENZT -> Gemini (gemini/) → Gemini 2.5 Flash – 1.500 Req/Tag kostenlos -> Cloudflare AI (cf/) → 50+ Modelle – 10.000 Neuronen/Tag -> Scaleway (scw/) → Qwen3 235B, Llama 70B – 1 Mio. kostenlose Token (EU) -> Groq (groq/) → Lama/Gemma – 14,4K req/Tag ultraschnell -> NVIDIA NIM (nvidia/) → 70+ offene Modelle – 40 U/min für immer -> Großhirn (Großhirn) → Lama/Qwen weltweit am schnellsten – 1 Mio. tok/Tag -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Transkribieren Sie jedes Audio/Video für**0 $**– Deepgram führt mit 200 $ kostenlos, AssemblyAI 50 $ Fallback, Groq Whisper als unbegrenztes Notfall-Backup. +## 🎙️ Free Transcription Combo -| Anbieter | Kostenlose Credits | Bestes Modell | Ratenlimit | -| ----------------- | ---------------------- | -------------------------------------------- | ------------- | -| 🟢**Deepgram**|**200 $ gratis**(Anmeldung) | „nova-3“ – beste Genauigkeit, über 30 Sprachen | Kein RPM-Limit für kostenlose Credits | -| 🔵**AssemblyAI**|**50 $ gratis**(Anmeldung) | „universal-3-pro“ – Kapitel, Stimmung, PII | Kein RPM-Limit für kostenlose Credits | -| 🔴**Groq**|**Für immer kostenlos**| „whisper-large-v3“ – OpenAI Whisper | 30 U/min (Geschwindigkeit begrenzt) | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. -**Vorgeschlagene Kombination in „/dashboard/combos“:**``` +| Provider | Free Credits | Best Model | Rate Limit | +| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | + +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Dann unter „/dashboard/media“ → Registerkarte „Transkription“: Laden Sie eine beliebige Audio- oder Videodatei hoch → wählen Sie Ihren Kombinationsendpunkt aus → erhalten Sie Transkriptionen in unterstützten Formaten.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 ist als Betriebsplattform konzipiert und nicht nur als Relay-Proxy.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Funktion | Was es tut | -| ------------------------------------- | --------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Grok-4 Fast Family** | xAI-Modelle für 0,20 $/0,50 $/M – im Benchmarking 1143 ms (30 % schneller als Gemini 2.5 Flash) | -| 🧠**GLM-5 über Z.AI** | 128K-Ausgabekontext, 0,5 $/1 Mio. – neuestes Flaggschiff der GLM-Familie | -| 🔮**MiniMax M2.5** | Argumentation + Agentenaufgaben für 0,30 $/1 Mio. – deutliche Verbesserung gegenüber M2.1 | -| 🎯**toolCalling Flag pro Modell** | Pro Modell „toolCalling: true/false“ in der Registrierung – AutoCombo überspringt nicht-toolfähige Modelle | -| 🌍**Mehrsprachige Absichtserkennung** | PT/ZH/ES/AR-Schlüsselwörter in der AutoCombo-Bewertung – bessere Modellauswahl für nicht-englische Inhalte | -| 📊**Benchmark-gesteuerte Fallbacks** | Echte p95-Latenz aus der Kombinationsbewertung von Live-Anfrage-Feeds – AutoCombo lernt aus tatsächlichen Daten | -| 🔁**Deduplizierung anfordern** | Content-Hash-basiertes Dedup-Fenster – Multi-Agent-sicher, verhindert doppelte Gebühren | -| 🔌**Pluggable RouterStrategy** | Erweiterbare „RouterStrategy“-Schnittstelle – benutzerdefinierte Routing-Logik als Plugins hinzufügen | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Funktion | Was es tut | -| ---------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Modellspielplatz** | Dashboard-Seite zum direkten Testen jedes Modells – Anbieter-/Modell-/Endpunkt-Selektoren, Monaco-Editor, Streaming, Abbruch, Timing | -| 🔏**CLI-Fingerabdruckabgleich** | Header-/Body-Reihenfolge pro Anbieter, um mit nativen CLI-Signaturen übereinzustimmen – schalten Sie pro Anbieter unter „Einstellungen“ > „Sicherheit“ um.**Ihre Proxy-IP bleibt erhalten** | -| 🤝**ACP-Unterstützung (Agent Client Protocol)** | CLI-Agent-Erkennung (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 weitere), Prozess-Spawner, „/api/acp/agents“-Endpunkt | -| 🤖**ACP-Agenten-Dashboard** | Debuggen › Seite „Agenten“ – Raster mit 14 Agenten mit Installationsstatus, Version und benutzerdefiniertem Agentenformular für jedes CLI-Tool.**OpenCode**-Benutzer erhalten eine Schaltfläche „Opencode.json herunterladen“, die automatisch eine gebrauchsfertige Konfiguration mit allen verfügbaren Modellen generiert. | -| 🔧**Benutzerdefiniertes Modell „apiFormat“-Routing** | Benutzerdefinierte Modelle mit „apiFormat: „responses““ werden jetzt korrekt an den Responses-API-Übersetzer weitergeleitet | -| 🏢**Codex Workspace Isolation** | Mehrere Codex-Arbeitsbereiche pro E-Mail – OAuth trennt Verbindungen korrekt nach Arbeitsbereichs-ID | -| 🔄**Electron Auto-Update** | Desktop-App sucht nach Updates + automatische Installation beim Neustart | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Funktion | Was es tut | -| ------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**MCP-Server (25 Tools)** | IDE/Agent-Tools über 3 Transporte: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 Kerne + 3 Speicher + 4 Fertigkeitswerkzeuge | -| 🤝**A2A-Server (JSON-RPC + SSE)** | Ausführung von Agent-zu-Agent-Aufgaben mit Synchronisierungs- und Streaming-Flows | -| 🧭**Consolidated Endpoints-Seite** | Verwaltungsseite mit Registerkarten mit den Registerkarten „Endpunkt-Proxy“, „MCP“, „A2A“ und „API-Endpunkte“ | -| 🎚️**Service-Aktivierung/Deaktivierung** | EIN/AUS-Schalter für MCP und A2A mit Einstellungspersistenz (Standard: AUS) | -| 🛰️**MCP Runtime Heartbeat** | Echter Prozessstatus (PID, Betriebszeit, Heartbeat-Alter, Transport, Scope-Modus) | -| 📋**MCP Audit Trail** | Filterbare Audit-Protokolle mit Erfolg/Misserfolg und Schlüsselzuordnung | -| 🔐**Durchsetzung des MCP-Geltungsbereichs** | 10 granulare Umfangsberechtigungen für kontrollierten Werkzeugzugriff | -| 📡**A2A Task Lifecycle Management** | Aufgaben auflisten/filtern, Ereignisse/Artefakte prüfen, laufende Aufgaben abbrechen | -| 📋**Agentenkartenerkennung** | `/.well-known/agent.json` für die automatische Client-Erkennung | -| 🧪**Protokoll-E2E-Testkabel** | Echtes MCP SDK + A2A-Client fließt in „test:protocols:e2e“ | -| ⚙️**Betriebskontrollen** | Schaltkombination, Anwenden von Resilienzprofilen, Zurücksetzen von Leistungsschaltern über eine Bedienoberfläche | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Funktion | Was es tut | -| --------------------------------------- | ---------------------------------------------------------------------------------------- | ----------------------- | -| 🎯**Intelligenter 4-Stufen-Fallback** | Automatische Route: Abonnement → API-Schlüssel → Günstig → Kostenlos | -| 📊**Kontingentverfolgung in Echtzeit** | Live-Token-Zählung + Reset-Countdown pro Anbieter | -| 🔄**Formatübersetzung** | OpenAI ↔ Claude ↔ Gemini ↔ Antworten mit schemasicheren Konvertierungen | -| 👥**Unterstützung mehrerer Konten** | Mehrere Konten pro Anbieter mit intelligenter Auswahl | -| 🔄**Automatische Token-Aktualisierung** | OAuth-Token werden bei Wiederholung automatisch aktualisiert | -| 🎨**Benutzerdefinierte Kombinationen** | 9 Ausgleichsstrategien + Fallback-Kettenkontrolle | -| 🌐**Wildcard-Router** | `provider/*` dynamisches Routing | -| 🧠**Budgetkontrollen denken** | Passthrough-, automatische, benutzerdefinierte und adaptive Reasoning-Grenzwerte | -| 🔀**Modell-Aliase** | Integrierte + benutzerdefinierte Modell-Aliasing- und Migrationssicherheit | -| ⚡**Hintergrundverschlechterung** | Hintergrundaufgaben mit niedriger Priorität an günstigere Modelle weiterleiten | -| 🧪**Aufgabenbewusstes Smart Routing** | Modell automatisch nach Inhaltstyp auswählen (Codierung/Vision/Analyse/Zusammenfassung) | -| 🔄**A2A-Agent-Workflows** | Deterministischer FSM-Orchestrator für zustandsbehaftete mehrstufige Agentenausführungen | -| 🔀**Adaptives Routing** | Dynamische Strategieüberschreibung basierend auf Token-Volumen und Prompt-Komplexität | -| 🎲**Anbietervielfalt** | Shannon-Entropiebewertung, die die Verteilung des Auto-Combo-Verkehrs ausgleicht | -| 💬**System-Prompt-Injektion** | Globale Verhaltenskontrollen werden konsequent angewendet | -| 📄**Antwort-API-Kompatibilität** | Vollständige „/v1/responses“-Unterstützung für Codex und erweiterte Agenten-Workflows | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Funktion | Was es tut | -| ------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ---------------------------------------- | -| 🖼️**Bilderzeugung** | `/v1/images/generations` mit Cloud- und lokalen Backends | -| 📐**Einbettungen** | `/v1/embeddings` für Such- und RAG-Pipelines | -| 🎤**Audio-Transkription** | „/v1/audio/transcriptions“ – 7 Anbieter (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), automatische Spracherkennung, MP4/MP3/WAV-Unterstützung | -| 🔊**Text-to-Speech** | „/v1/audio/speech“ – 10 Anbieter (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) mit korrekten Fehlermeldungen | -| 🎬**Videogenerierung** | `/v1/videos/generations` (ComfyUI + SD WebUI-Workflows) | -| 🎵**Musikgeneration** | `/v1/music/generations` (ComfyUI-Workflows) | -| 🛡️**Moderationen** | `/v1/moderations` Sicherheitsüberprüfungen | -| 🔀**Neueinstufung** | `/v1/rerank` für Relevanzbewertung | -| 🔍**Websuche**🆕 | „/v1/search“ – 5 Anbieter (Serper, Brave, Perplexity, Exa, Tavily), 6.500+ kostenlos/Monat, automatisches Failover, Cache | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Funktion | Was es tut | -| ----------------------------------------------- | ------------------------------------------------------------------------------------------------------------------ | -------------------------------- | -| 🔌**Leistungsschalter** | Auslösung/Wiederherstellung pro Modell mit Schwellenwertkontrollen | -| 🎯**Endpunktfähige Modelle** | Benutzerdefinierte Modelle deklarieren unterstützte Endpunkte + API-Format | -| 🛡️**Anti-Donnerende Herde** | Mutex- und Semaphorschutz bei Wiederholungs-/Ratenereignissen | -| 🧠**Semantik + Signatur-Cache** | Kosten-/Latenzreduzierung mit zwei Cache-Schichten | -| ⚡**Idempotenz anfordern** | Doppeltes Schutzfenster | -| 🔒**TLS-Fingerabdruck-Spoofing** | Browserähnlicher TLS-Fingerabdruck –**reduziert die Bot-Erkennung und Kontokennzeichnung** | -| 🔏**CLI-Fingerabdruckabgleich** | Entspricht nativen CLI-Anfragesignaturen –**reduziert das Verbotsrisiko und behält gleichzeitig die Proxy-IP bei** | -| 🌐**IP-Filterung** | Zulassungs-/Blocklistenkontrolle für exponierte Bereitstellungen | -| 📊**Bearbeitbare Ratenlimits** | Konfigurierbare globale/Provider-Level-Limits mit Persistenz | -| 📉**Anmutige Degradierung** | Mehrschichtige Fallbacks zum Schutz des Kern-Gateway-Betriebs | -| 📜**Audit-Trail konfigurieren** | Diff-basierte Änderungsverfolgung verhindert betriebliche Abweichungen durch einfache Rollbacks | -| ⏳**Provider Health Sync** | Proaktive Überwachung des Token-Ablaufs, die Warnungen vor Autorisierungsfehlern auslöst | -| 🚪**Gesperrte Konten automatisch deaktivieren** | Funktionsfähiger Leistungsschalter, der dauerhaft gesperrte Token-Konten automatisch verschließt | -| 🔑**API-Schlüsselverwaltung + Scoping** | Sichere Schlüsselausgabe/-rotation und Modell-/Anbieterkontrollen | -| 👁️**Scoped API Key Reveal**🆕 | Opt-in-Wiederherstellung von API-Schlüsseln über „ALLOW_API_KEY_REVEAL“ | -| 🛡️**Geschützte „/Modelle“** | Optionales Authentifizierungs-Gating und Provider-Ausblenden für Modellkatalog | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Funktion | Was es tut | -| ------------------------------------------ | ------------------------------------------------------------------- | ---------------------------- | -| 📝**Anfrage + Proxy-Protokollierung** | Vollständige Anfrage/Antwort- und Proxy-Protokollierung | -| 📉**Gestreamte detaillierte Protokolle**🆕 | Rekonstruiert SSE-Nutzlastströme sauber in der Benutzeroberfläche | -| 📋**Einheitliches Protokoll-Dashboard** | Anforderungs-, Proxy-, Audit- und Konsolenansichten auf einer Seite | -| 🔍**Telemetrie anfordern** | p50/p95/p99-Latenz und Anforderungsverfolgung | -| 🏥**Gesundheits-Dashboard** | Betriebszeit, Breaker-Zustände, Sperrungen, Cache-Statistiken | -| 💰**Kostenverfolgung** | Budgetkontrolle und Preistransparenz pro Modell | -| 📈**Analysevisualisierungen** | Einblicke in die Modell-/Anbieternutzung und Trendansichten | -| 🧪**Bewertungsrahmen** | Golden-Set-Test mit konfigurierbaren Match-Strategien | -| 📡**Live-Diagnose**🆕 | Semantische Cache-Umgehung für genaue Combo-Live-Tests | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Funktion | Was es tut | -| ------------------------------------------ | ------------------------------------------------------------------------------ | --------------------- | -| 🌐**Überall bereitstellen** | Localhost, VPS, Docker, Cloud-Umgebungen | -| 🚇**Cloudflare-Tunnel**🆕 | Quick-Tunnel-Integration mit einem Klick über das Dashboard | -| 🔑**API-Schlüsselmodellfilterung** | Native /v1/models-Antwort gefiltert über zugewiesene Bearer-Kontextrollen | -| ⚡**Smart Cache Bypass** | Konfigurierbare TTL-Heuristik und erzwungene Refetch-Kontrollen | -| 🔄**Sichern/Wiederherstellen** | Export-/Import- und Disaster-Recovery-Abläufe | -| 🧙**Onboarding-Assistent** | Erstmaliges geführtes Setup | -| 🔧**CLI-Tools-Dashboard** | Ein-Klick-Setup für beliebte Codierungstools | -| 🎮**Modellspielplatz** | Testen Sie alle Anbieter/Modelle/Endpunkte über das Dashboard | -| 🔏**CLI-Fingerabdruck-Umschaltung** | Fingerabdruckabgleich pro Anbieter unter Einstellungen > Sicherheit | -| 🌐**i18n (30 Sprachen)** | Vollständige Sprachunterstützung für Dashboard und Dokumente mit RTL-Abdeckung | -| 🧹**Alle Modelle löschen** | Löschen der Modellliste in den Anbieterdetails mit einem Klick | -| 👁️**Sidebar-Steuerelemente**🆕 | Komponenten und Integrationen in den Darstellungseinstellungen ausblenden | -| 📋**Problemvorlagen** | Standardisierte GitHub-Vorlagen für Fehler und Funktionen | -| 📂**Benutzerdefiniertes Datenverzeichnis** | „DATA_DIR“-Überschreibung für Speicherort | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1295,103 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -Wenn Kontingent, Rate oder Integrität fehlschlagen, wechselt OmniRoute automatisch zum nächsten Kandidaten, ohne dass ein manueller Wechsel erforderlich ist.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A sind in der Benutzeroberfläche und in den Dokumenten erkennbar (nicht ausgeblendet) -- Protokollstatus-APIs stellen Live-Betriebsdaten bereit (`/api/mcp/*`, `/api/a2a/*`) -- Dashboards umfassen Aktionen für Tag-2-Operationen (Kombinationsumschaltung, Zurücksetzen von Leistungsschaltern, Aufgabenabbruch).#### Translator + validation workflow +#### Protocol management that is visible and operable -Der Übersetzerbereich umfasst: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Spielplatz**: Transformationsprüfungen anfordern -**Chat-Tester**: vollständiger Anfrage-/Antwort-Roundtrip -**Prüfstand**: mehrere Fälle in einem Durchgang -**Live Monitor**: Echtzeit-Verkehrsansicht +#### Translator + validation workflow -Plus Protokollvalidierung mit echten Clients über „npm run test:protocols:e2e“. +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**– Tool-Referenz, IDE-Konfigurationen und Client-Beispiele +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A Server README](src/lib/a2a/README.md)**– Fähigkeiten, JSON-RPC-Methoden, Streaming und Aufgabenlebenszyklus## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute umfasst ein integriertes Bewertungsframework zum Testen der LLM-Antwortqualität anhand eines Golden Sets. Greifen Sie darauf über**Analytics → Evals**im Dashboard zu.### Built-in Golden Set +## 🧪 Evaluations (Evals) -Das vorinstallierte „OmniRoute Golden Set“ enthält Testfälle für: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Grüße, Mathematik, Geographie, Codegenerierung -- Einhaltung des JSON-Formats, Übersetzung, Markdown-Generierung -- Sicherheitsverweigerung (schädlicher Inhalt), Zählung, boolesche Logik### Evaluation Strategies +### Built-in Golden Set -| Strategie | Beschreibung | Beispiel | -| ------------------- | -------------------------------------------------------------------------------------------- | --------------------------------------- | --- | -| „genau“ | Die Ausgabe muss genau mit | übereinstimmen „4“ | -| „enthält“ | Die Ausgabe muss eine Teilzeichenfolge enthalten (Groß-/Kleinschreibung wird nicht beachtet) | „Paris“ | -| `regex` | Die Ausgabe muss mit dem Regex-Muster | übereinstimmen `"1.*2.*3"` | -| „Benutzerdefiniert“ | Benutzerdefinierte JS-Funktion gibt true/false | zurück `(Ausgabe) => Ausgabelänge > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) -
-🧩 MCP-Setup (Model Context Protocol) +
+🧩 MCP Setup (Model Context Protocol) -Starten Sie den MCP-Transport im Standardmodus:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Empfohlener Validierungsablauf: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Verbinden Sie Ihren MCP-Client über stdio. -2. Führen Sie „omniroute_get_health“ aus. -3. Führen Sie „omniroute_list_combos“ aus. -4. Öffnen Sie „/dashboard/mcp“, um Heartbeat, Aktivität und Audit zu bestätigen. - -Nützliche APIs für die Automatisierung: +Useful APIs for automation: - `GET /api/mcp/status` - `GET /api/mcp/tools` - `GET /api/mcp/audit` -- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats` -
-🤝 A2A-Setup (Agent2Agent) +
-Entdecken Sie den Agenten:```bash +
+🤝 A2A Setup (Agent2Agent) + +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Senden Sie eine Aufgabe:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` - -Lebenszyklus verwalten: +Manage lifecycle: - `GET /api/a2a/status` - `GET /api/a2a/tasks` - `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -Operative Benutzeroberfläche: +Operational UI: -- „/dashboard/a2a“ für Aufgaben-/Status-/Stream-Beobachtbarkeit und Smoke-Aktionen
+- `/dashboard/a2a` for task/state/stream observability and smoke actions -
-🧪 End-to-End-Protokollvalidierung +
-Validieren Sie beide Protokolle mit echten Clients:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Dies bestätigt: +This verifies: -- MCP SDK-Client-Verbindung/Liste/Anruf -- A2A-Erkennung/Senden/Streamen/Get/Abbrechen -- Vergleichen Sie die Daten in MCP-Audit- und A2A-Aufgabenverwaltungs-APIs
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs -
-💳 Abonnementanbieter### Claude Code (Pro/Max) +
+ +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1404,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Profi-Tipp:**Verwenden Sie Opus für komplexe Aufgaben, Sonnet für Geschwindigkeit. OmniRoute verfolgt das Kontingent pro Modell!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1418,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Für jedes Codex-Konto gibt es jetzt Richtlinienumschaltungen unter „Dashboard -> Anbieter“: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- „5h“ (EIN/AUS): Erzwingt die 5-Stunden-Fensterschwellenrichtlinie. -- „Wöchentlich“ (EIN/AUS): Erzwingen Sie die wöchentliche Fensterschwellenrichtlinie. - – Schwellenwertverhalten: Wenn ein aktiviertes Fenster eine Nutzung von >=90 % erreicht, wird dieses Konto übersprungen. -- Rotationsverhalten: OmniRoute leitet automatisch zum nächsten berechtigten Codex-Konto weiter. -- Zurücksetzungsverhalten: Wenn die „resetAt“-Zeit des Anbieters verstrichen ist, wird das Konto automatisch wieder berechtigt. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Szenarien: +Scenarios: -- „5 Stunden EIN“ + „Wöchentlich EIN“: Das Konto wird übersprungen, wenn eines der Fenster den Schwellenwert erreicht. -- „5h AUS“ + „Wöchentlich EIN“: Nur wöchentliche Nutzung kann das Konto sperren. -- „5h EIN“ + „Wöchentlich AUS“: Nur eine 5-stündige Nutzung kann das Konto sperren. -- „resetAt“ übergeben: Das Konto wechselt automatisch wieder in die Rotation (keine manuelle erneute Aktivierung).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1443,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Bester Wert:**Riesiges kostenloses Kontingent! Verwenden Sie dies vor kostenpflichtigen Stufen.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1458,71 +1662,91 @@ Models:
-
-🔑 API-Schlüsselanbieter### NVIDIA NIM (FREE developer access — 70+ models) +
+🔑 API Key Providers -1. Registrieren Sie sich: [build.nvidia.com](https://build.nvidia.com) -2. Holen Sie sich einen kostenlosen API-Schlüssel (1000 Inferenz-Credits inbegriffen) -3. Dashboard → Anbieter hinzufügen → NVIDIA NIM: - - API-Schlüssel: „nvapi-your-key“. +### NVIDIA NIM (FREE developer access — 70+ models) -**Modelle:**„nvidia/llama-3.3-70b-instruct“, „nvidia/mistral-7b-instruct“ und mehr als 50 weitere +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Profi-Tipp:**OpenAI-kompatible API – funktioniert nahtlos mit der Formatübersetzung von OmniRoute!### DeepSeek +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -1. Registrieren Sie sich: [platform.deepseek.com](https://platform.deepseek.com) -2. Holen Sie sich den API-Schlüssel -3. Dashboard → Anbieter hinzufügen → DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -**Modelle:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +### DeepSeek -1. Registrieren Sie sich: [console.groq.com](https://console.groq.com) -2. Holen Sie sich den API-Schlüssel (kostenloses Kontingent inbegriffen) -3. Dashboard → Anbieter hinzufügen → Groq +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -**Modelle:**„groq/llama-3.3-70b“, „groq/mixtral-8x7b“. +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Profi-Tipp:**Ultraschnelle Inferenz – am besten für Echtzeit-Codierung!### OpenRouter (100+ Models) +### Groq (Free Tier Available!) -1. Registrieren Sie sich: [openrouter.ai](https://openrouter.ai) -2. Holen Sie sich den API-Schlüssel -3. Dashboard → Anbieter hinzufügen → OpenRouter +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -**Modelle:**Greifen Sie über einen einzigen API-Schlüssel auf über 100 Modelle aller großen Anbieter zu. +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Dashboard-Verhalten:**OpenRouter-Modelle werden über**Verfügbare Modelle**verwaltet. Durch manuelles Hinzufügen, Importieren und automatische Synchronisieren wird dieselbe Liste aktualisiert.
+**Pro Tip:** Ultra-fast inference — best for real-time coding! -
-💰 Günstige Anbieter (Backup)### GLM-4.7 (Daily reset, $0.6/1M) +### OpenRouter (100+ Models) -1. Registrieren Sie sich: [Zhipu AI](https://open.bigmodel.cn/) -2. Holen Sie sich den API-Schlüssel vom Coding Plan -3. Dashboard → API-Schlüssel hinzufügen: - - Anbieter: `glm` - - API-Schlüssel: „Ihr-Schlüssel“. +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -**Verwenden Sie:**`glm/glm-4.7` +**Models:** Access 100+ models from all major providers through a single API key. -**Profi-Tipp:**Coding Plan bietet 3× Kontingent zu 1/7 Kosten! Täglich um 10:00 Uhr zurückgesetzt.### MiniMax M2.1 (5h reset, $0.20/1M) +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -1. Registrieren Sie sich: [MiniMax](https://www.minimax.io/) -2. Holen Sie sich den API-Schlüssel -3. Dashboard → API-Schlüssel hinzufügen +
-**Verwenden Sie:**„minimax/MiniMax-M2.1“. +
+💰 Cheap Providers (Backup) -**Profi-Tipp:**Günstigste Option für langen Kontext (1 Mio. Token)!### Kimi K2 ($9/month flat) +### GLM-4.7 (Daily reset, $0.6/1M) -1. Abonnieren: [Moonshot AI](https://platform.moonshot.ai/) -2. Holen Sie sich den API-Schlüssel -3. Dashboard → API-Schlüssel hinzufügen +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**Verwendung:**`kimi/kimi-latest` +**Use:** `glm/glm-4.7` -**Profi-Tipp:**Festpreis: 9 $/Monat für 10 Mio. Token = 0,90 $/1 Mio. effektive Kosten!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. -
-🆓 KOSTENLOSE Anbieter (Notfall-Backup)### Qoder (5 FREE models via OAuth) +### MiniMax M2.1 (5h reset, $0.20/1M) + +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `minimax/MiniMax-M2.1` + +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + +
+ +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1563,8 +1787,10 @@ Models:
-
-🎨 Combos erstellen### Example 1: Maximize Subscription → Cheap Backup +
+🎨 Create Combos + +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1592,8 +1818,10 @@ Cost: $0 forever!
-
-🔧 CLI-Integration### Cursor IDE +
+🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1604,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Verwenden Sie die Seite**CLI-Tools**im Dashboard für die Ein-Klick-Konfiguration oder bearbeiten Sie „~/.claude/settings.json“ manuell.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1615,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Option 1 – Dashboard (empfohlen):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Option 2 – Manuell:**Bearbeiten Sie „~/.openclaw/openclaw.json“:```json +```json { "models": { "providers": { @@ -1632,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Hinweis:**OpenClaw funktioniert nur mit lokaler OmniRoute. Verwenden Sie „127.0.0.1“ anstelle von „localhost“, um Probleme mit der IPv6-Auflösung zu vermeiden.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1646,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Schritt 1:**OmniRoute als benutzerdefinierten Anbieter hinzufügen:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Schritt 2:**Erstellen/bearbeiten Sie „opencode.json“ in Ihrem Projektstammverzeichnis:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1672,117 +1909,130 @@ opencode } } } -```` +``` -**Schritt 3:**Wählen Sie das Modell in OpenCode aus:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Tipp:**Fügen Sie alle in Ihrem OmniRoute-Endpunkt „/v1/models“ verfügbaren Modelle zum Abschnitt „Modelle“ hinzu. Verwenden Sie das Format „Anbieter/Modell-ID“ aus Ihrem OmniRoute-Dashboard.
+
--- ## Fehlerbehebung -
-Klicken Sie hier, um die Anleitung zur Fehlerbehebung zu erweitern +
+Click to expand troubleshooting guide -**„Sprachmodell hat keine Nachrichten bereitgestellt“** +**"Language model did not provide messages"** -- Anbieterkontingent erschöpft → Überprüfen Sie den Dashboard-Kontingent-Tracker -- Lösung: Combo-Fallback verwenden oder auf günstigere Stufe wechseln +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**Ratenbegrenzung** +**Rate limiting** -- Abonnementkontingent aufgebraucht → Fallback auf GLM/MiniMax -- Kombination hinzufügen: „cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking“. +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**OAuth-Token abgelaufen** +**OAuth token expired** -- Automatische Aktualisierung durch OmniRoute -- Wenn die Probleme weiterhin bestehen: Dashboard → Anbieter → Verbindung wiederherstellen +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Hohe Kosten** +**High costs** -- Überprüfen Sie die Nutzungsstatistiken im Dashboard → Kosten -- Primärmodell auf GLM/MiniMax umstellen -- Nutzen Sie den kostenlosen Tarif (Gemini CLI, Qoder) für unkritische Aufgaben +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Dashboard-/API-Ports sind falsch** +**Dashboard/API ports are wrong** -- „PORT“ ist der kanonische Basisport (und standardmäßig API-Port) -– „API_PORT“ überschreibt nur den OpenAI-kompatiblen API-Listener -– „DASHBOARD_PORT“ überschreibt nur den Dashboard/Next.js-Listener -- Setzen Sie „NEXT_PUBLIC_BASE_URL“ auf Ihr Dashboard/öffentliche URL (für OAuth-Rückrufe) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Cloud-Synchronisierungsfehler** +**Cloud sync errors** -- Überprüfen Sie, ob „BASE_URL“ auf Ihre laufende Instanz verweist -– Überprüfen Sie, ob „CLOUD_URL“ auf Ihren erwarteten Cloud-Endpunkt verweist -- Halten Sie die Werte von „NEXT_PUBLIC_*“ an den serverseitigen Werten ausgerichtet +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Erste Anmeldung funktioniert nicht** +**First login not working** -- Überprüfen Sie „INITIAL_PASSWORD“ in „.env“. -- Wenn nicht festgelegt, lautet das Fallback-Passwort „123456“. +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Keine Anfrageprotokolle** +**No request logs** -– Anforderungsartefakte werden als eine JSON-Datei pro Anforderung in „DATA_DIR/call_logs/“ geschrieben -- Aktivieren Sie die Pipeline-Erfassung über Dashboard → Protokolle → Protokolle anfordern, wenn Sie detaillierte Payloads pro Phase benötigen -- Legen Sie „APP_LOG_TO_FILE=true“ fest, wenn Sie auch Anwendungskonsolenprotokolle in „logs/application/app.log“ haben möchten -- Passen Sie „APP_LOG_MAX_FILE_SIZE“, „APP_LOG_RETENTION_DAYS“, „APP_LOG_MAX_FILES“ und „CALL_LOG_MAX_ENTRIES“ nach Bedarf an +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Verbindungstest zeigt „Ungültig“ für OpenAI-kompatible Anbieter** +**Connection test shows "Invalid" for OpenAI-compatible providers** -– Viele Anbieter stellen keinen „/models“-Endpunkt bereit -– OmniRoute v1.0.6+ beinhaltet eine Fallback-Validierung über Chat-Abschlüsse -– Stellen Sie sicher, dass die Basis-URL das Suffix „/v1“ enthält### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server - + ->**⚠️ Wichtig für Benutzer, die OmniRoute auf einem VPS, Docker oder einem anderen Remote-Server ausführen**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -Die Anbieter**Antigravity**und**Gemini CLI**verwenden**Google OAuth 2.0**. Google verlangt, dass „redirect_uri“ im OAuth-Flow genau mit einem der vorregistrierten URIs in der Google Cloud Console der App übereinstimmt. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -Die in OmniRoute gebündelten OAuth-Anmeldeinformationen werden**nur für „localhost“**registriert. Wenn Sie auf OmniRoute auf einem Remote-Server zugreifen (z. B. „https://omniroute.myserver.com“), lehnt Google die Authentifizierung mit Folgendem ab:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Sie müssen in der Google Cloud Console eine**OAuth 2.0-Client-ID**mit dem URI Ihres Servers erstellen.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Öffnen Sie die Google Cloud Console** +#### Step-by-step -Gehen Sie zu: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Erstellen Sie eine neue OAuth 2.0-Client-ID** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Klicken Sie auf**„+ Anmeldeinformationen erstellen“**→**„OAuth-Client-ID“** -- Anwendungstyp:**„Webanwendung“** -- Name: beliebig (z. B. „OmniRoute Remote“) +**2. Create a new OAuth 2.0 Client ID** -**3. Autorisierte Weiterleitungs-URIs hinzufügen** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -Fügen Sie im Feld**"Autorisierte Weiterleitungs-URIs"**Folgendes hinzu:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Ersetzen Sie „Ihr-Server.com“ durch die Domäne oder IP Ihres Servers (geben Sie bei Bedarf den Port ein, z. B. „http://45.33.32.156:20128/callback“). +**4. Save and copy the credentials** -**4. Speichern und kopieren Sie die Anmeldeinformationen** +After creating, Google will show the **Client ID** and **Client Secret**. -Nach der Erstellung zeigt Google die**Client-ID**und das**Client-Geheimnis**an. +**5. Set environment variables** -**5. Umgebungsvariablen festlegen** +In your `.env` (or Docker environment variables): -In Ihrer „.env“ (oder Docker-Umgebungsvariablen):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1791,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. OmniRoute neu starten**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Versuchen Sie erneut, eine Verbindung herzustellen** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Dashboard → Anbieter → Antigravity (oder Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google leitet jetzt korrekt zu „https://your-server.com/callback“ weiter.--- +--- #### Temporary workaround (without custom credentials) -Wenn Sie jetzt keine eigenen Anmeldeinformationen einrichten möchten, können Sie dennoch den**manuellen URL-Ablauf**verwenden: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute öffnet die Google-Autorisierungs-URL -2. Nach der Autorisierung versucht Google, auf „localhost“ umzuleiten (was auf dem Remote-Server fehlschlägt). -3.**Kopieren Sie die vollständige URL**aus der Adressleiste Ihres Browsers (auch wenn die Seite nicht geladen wird) -4. Fügen Sie diese URL in das Feld ein, das im OmniRoute-Verbindungsmodal angezeigt wird -5. Klicken Sie auf**„Verbinden“** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Dies funktioniert, weil der Autorisierungscode in der URL unabhängig davon gültig ist, ob die Weiterleitungsseite geladen wurde.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. -
-🇧🇷 Versão em Português#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Wir haben**Antigravity**und**Gemini CLI**mit**Google OAuth 2.0**zur Authentifizierung getestet. Google erwartet, dass „redirect_uri“ kein OAuth-Fluss verwendet, da**exatamente**ein URI vorab in die Google Cloud Console aufgenommen wurde. +
+🇧🇷 Versão em Português -Als OAuth-Anmelder wurde OmniRoute nicht als „localhost“**registriert. Wenn Sie auf einen Remote-Server (z. B. „https://omniroute.meuservidor.com“) auf OmniRoute zugreifen, lehnt Google die Authentifizierung ab:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Sie schreiben bitte eine**OAuth 2.0-Client-ID**in der Google Cloud Console mit einem URI für Ihren Server.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Zugriff auf die Google Cloud Console** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -**2. Rufen Sie eine neue OAuth 2.0-Client-ID auf** +**2. Crie um novo OAuth 2.0 Client ID** -- Klicken Sie auf**"+ Anmeldeinformationen erstellen"**→**"OAuth-Client-ID"** -- Anwendungstyp:**„Webanwendung“** -- Name: Wählen Sie einen beliebigen Namen (z. B. „OmniRoute Remote“) +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Adicione als autorisierte Weiterleitungs-URIs** +**3. Adicione as Authorized Redirect URIs** -Nein,**"Autorisierte Weiterleitungs-URIs"**, Zusatz:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Ersetzen Sie Ihren Server durch „seu-servidor.com“ oder die IP Ihres Servers (einschließlich der erforderlichen Portierung, z. B. „http://45.33.32.156:20128/callback“). +**4. Salve e copie as credenciais** -**4. Als Anmeldedaten speichern und kopieren** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Anschließend hat Google die**Client-ID**und das**Client-Geheimnis**angezeigt. +**5. Configure as variáveis de ambiente** -**5. Als Umgebungsvariationen konfigurieren** +No seu `.env` (ou nas variáveis de ambiente do Docker): -Kein `.env` (oder mehrere Docker-Umgebungsvarianten):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1870,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Neuzugang zu OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute - -```` +``` **7. Tente conectar novamente** -Dashboard → Anbieter → Antigravity (oder Gemini CLI) → OAuth +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Dann leiten Sie Google direkt an „https://seu-servidor.com/callback“ weiter und überprüfen Sie die Funktion.--- +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. + +--- #### Workaround temporário (sem configurar credenciais próprias) -Wenn Sie vorab keine Berechtigung erhalten möchten, besteht die Möglichkeit, das**URL-Handbuch**zu verwenden: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. OmniRoute ruft eine von Google autorisierte URL auf -2. Nachdem Sie den Autor autorisiert haben, sendet Google eine Weiterleitung an „localhost“ (das bedeutet, dass Sie den Server nicht weiterleiten können). -3.**Kopieren Sie eine vollständige URL**, um sie in Ihren Browser zu laden (bitte beachten Sie, dass die Seite noch nicht abgeschlossen ist). -4. Geben Sie die URL ein, die nicht zur Verbindung mit OmniRoute verwendet werden soll -5. Klicken Sie auf**„Connect“** +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) +4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute +5. Clique em **"Connect"** -> Diese Problemumgehung funktioniert aufgrund des Autorisierungscodes auf der URL und ist unabhängig von der Weiterleitung oder Nicht-Weiterleitung gültig.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1908,64 +2171,73 @@ Wenn Sie vorab keine Berechtigung erhalten möchten, besteht die Möglichkeit, d ## 🛠️ Tech Stack -
-Klicken Sie hier, um die Tech-Stack-Details zu erweitern +
+Click to expand tech stack details --**Laufzeit**: Node.js 18–22 LTS (⚠️ Node.js 24+ wird**nicht unterstützt**– native Binärdateien von „better-sqlite3“ sind inkompatibel) --**Sprache**: TypeScript 5.9 –**100 % TypeScript**über „src/“ und „open-sse/“ (kein „any“ in Kernmodulen seit Version 2.0) --**Framework**: Next.js 16 + React 19 + Tailwind CSS 4 --**Datenbank**: LowDB (JSON) + SQLite (Domänenstatus + Proxy-Protokolle + MCP-Prüfung + Routing-Entscheidungen) --**Schemas**: Zod (MCP-Tool-I/O-Validierung, API-Verträge) --**Protokolle**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Streaming**: Vom Server gesendete Ereignisse (SSE) --**Auth**: OAuth 2.0 (PKCE) + JWT + API-Schlüssel + MCP-bezogene Autorisierung --**Testen**: Node.js-Testläufer + Vitest (über 900 Tests einschließlich Einheit, Integration, E2E) --**CI/CD**: GitHub-Aktionen (automatische NPM-Veröffentlichung + Docker Hub bei Veröffentlichung) --**Website**: [omniroute.online](https://omniroute.online) --**Paket**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Resilienz**: Leistungsschalter, exponentielles Backoff, Anti-Donner-Herde, TLS-Spoofing, automatische Kombinations-Selbstheilung
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + +
--- ## Dokumentation -| Dokument | Beschreibung | +| Document | Description | | ---------------------------------------------- | --------------------------------------------------- | -| [Benutzerhandbuch](docs/USER_GUIDE.md) | Anbieter, Kombinationen, CLI-Integration, Bereitstellung | -| [API-Referenz](docs/API_REFERENCE.md) | Alle Endpunkte mit Beispielen | -| [MCP-Server](open-sse/mcp-server/README.md) | 16 MCP-Tools, IDE-Konfigurationen, Python/TS/Go-Clients | -| [A2A-Server](src/lib/a2a/README.md) | JSON-RPC 2.0-Protokoll, Fähigkeiten, Streaming, Aufgabenverwaltung | -| [Auto-Combo-Engine](docs/auto-combo.md) | 6-Faktor-Bewertung, Moduspakete, Selbstheilung | -| [Fehlerbehebung](docs/TROUBLESHOOTING.md) | Häufige Probleme und Lösungen | -| [Architektur](docs/ARCHITECTURE.md) | Systemarchitektur und Interna | -| [Mitwirken](CONTRIBUTING.md) | Entwicklungsaufbau und Richtlinien | -| [OpenAPI-Spezifikation](docs/openapi.yaml) | OpenAPI 3.0-Spezifikation | -| [Sicherheitsrichtlinie](SECURITY.md) | Schwachstellenmeldung und Sicherheitspraktiken | -| [VM-Bereitstellung](docs/VM_DEPLOYMENT_GUIDE.md) | Vollständige Anleitung: VM + Nginx + Cloudflare-Setup | -| [Features-Galerie](docs/FEATURES.md) | Visuelle Dashboard-Tour mit Screenshots | -| [Release-Checkliste](docs/RELEASE_CHECKLIST.md) | Validierungsschritte vor der Veröffentlichung |--- +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -Für OmniRoute sind**210+ Funktionen**in mehreren Entwicklungsphasen geplant. Hier sind die Schlüsselbereiche: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Kategorie | Geplante Funktionen | Höhepunkte | -| -------------- | ---------------- | -------------------------------------------------------------------------------------- | -| 🧠**Routing & Intelligenz**| 25+ | Routing mit der niedrigsten Latenz, Tag-basiertes Routing, Quoten-Preflight, P2C-Kontoauswahl | -| 🔒**Sicherheit & Compliance**| 20+ | SSRF-Härtung, Credential-Cloaking, Ratenbegrenzung pro Endpunkt, Verwaltungsschlüssel-Scoping | -| 📊**Beobachtbarkeit**| 15+ | OpenTelemetry-Integration, Echtzeit-Kontingentüberwachung, Kostenverfolgung pro Modell | -| 🔄**Anbieterintegrationen**| 20+ | Dynamische Modellregistrierung, Anbieter-Abklingzeiten, Multi-Account-Codex, Copilot-Kontingentanalyse | -| ⚡**Leistung**| 15+ | Duale Cache-Schicht, Prompt-Cache, Antwort-Cache, Streaming-Keepalive, Batch-API | -| 🌐**Ökosystem**| 10+ | WebSocket-API, Hot-Reload der Konfiguration, verteilter Konfigurationsspeicher, kommerzieller Modus |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**OpenCode-Integration**– Native Anbieterunterstützung für die OpenCode AI-Codierungs-IDE -- 🔗**TRAE-Integration**– Volle Unterstützung für das TRAE AI-Entwicklungsframework -- 📦**Batch-API**– Asynchrone Stapelverarbeitung für Massenanfragen -- 🎯**Tag-basiertes Routing**– Leiten Sie Anfragen basierend auf benutzerdefinierten Tags und Metadaten weiter -- 💰**Niedrigste Kostenstrategie**– Wählen Sie automatisch den günstigsten verfügbaren Anbieter aus +### 🔜 Coming Soon -> 📝 Vollständige Funktionsspezifikationen verfügbar unter [`docs/new-features/`](docs/new-features/) (217 detaillierte Spezifikationen)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1973,18 +2245,20 @@ Für OmniRoute sind**210+ Funktionen**in mehreren Entwicklungsphasen geplant. Hi ### How to Contribute -1. Forken Sie das Repository -2. Erstellen Sie Ihren Feature-Zweig („git checkout -b feature/amazing-feature“) -3. Übernehmen Sie Ihre Änderungen („git commit -m ‚Erstaunliche Funktion hinzufügen‘“) -4. Zum Zweig pushen („git push origin feature/amazing-feature“) -5. Öffnen Sie eine Pull-Anfrage +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Detaillierte Richtlinien finden Sie unter [CONTRIBUTING.md](CONTRIBUTING.md).### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1996,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Besonderer Dank geht an**[9router](https://github.com/decolua/9router)**von**[decolua](https://github.com/decolua)**– das ursprüngliche Projekt, das diesen Fork inspiriert hat. OmniRoute baut auf dieser unglaublichen Grundlage mit zusätzlichen Funktionen, multimodalen APIs und einer vollständigen Neufassung von TypeScript auf. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Besonderer Dank geht an**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**– die ursprüngliche Go-Implementierung, die diese JavaScript-Portierung inspiriert hat.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Lizenz -MIT-Lizenz – Einzelheiten finden Sie unter [LIZENZ](LIZENZ).--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/de/docs/ARCHITECTURE.md b/docs/i18n/de/docs/ARCHITECTURE.md index 61a1ecca90..807f2a4df7 100644 --- a/docs/i18n/de/docs/ARCHITECTURE.md +++ b/docs/i18n/de/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Letzte Aktualisierung: 28.03.2026_## Executive Summary -OmniRoute ist ein lokales KI-Routing-Gateway und Dashboard, das auf Next.js basiert. -Es bietet einen einzigen OpenAI-kompatiblen Endpunkt („/v1/\*“) und leitet den Datenverkehr über mehrere Upstream-Anbieter mit Übersetzung, Fallback, Token-Aktualisierung und Nutzungsverfolgung weiter. -Kernkompetenzen: +_Last updated: 2026-03-28_ -- OpenAI-kompatible API-Oberfläche für CLI/Tools (28 Anbieter) -- Anforderungs-/Antwortübersetzung über Anbieterformate hinweg -- Modell-Combo-Fallback (Multi-Modell-Sequenz) -- Fallback auf Kontoebene (mehrere Konten pro Anbieter) -- OAuth + API-Schlüssel-Provider-Verbindungsverwaltung -- Einbettungsgenerierung über „/v1/embeddings“ (6 Anbieter, 9 Modelle) -- Bildgenerierung über „/v1/images/generations“ (4 Anbieter, 9 Modelle) -- Think-Tag-Parsing (`...`) für Argumentationsmodelle -- Antwortbereinigung für strikte OpenAI SDK-Kompatibilität -- Rollennormalisierung (Entwickler→System, System→Benutzer) für anbieterübergreifende Kompatibilität -- Strukturierte Ausgabekonvertierung (json_schema → Gemini ResponseSchema) -- Lokale Persistenz für Anbieter, Schlüssel, Aliase, Kombinationen, Einstellungen, Preise -- Nutzungs-/Kostenverfolgung und Anforderungsprotokollierung -- Optionale Cloud-Synchronisierung für die Synchronisierung mehrerer Geräte/Status -- IP-Zulassungs-/Blockierungsliste für die API-Zugriffskontrolle -- Denken Sie an die Budgetverwaltung (Passthrough/Auto/Benutzerdefiniert/Adaptiv) -- Sofortige Injektion des globalen Systems -- Sitzungsverfolgung und Fingerabdruck -- Erweiterte Ratenbegrenzung pro Konto mit anbieterspezifischen Profilen -- Leistungsschaltermuster für die Ausfallsicherheit des Anbieters -- Donnernder Herdenschutz mit Mutex-Sperre - – Signaturbasierter Anforderungsdeduplizierungs-Cache -- Domänenschicht: Modellverfügbarkeit, Kostenregeln, Fallback-Richtlinie, Sperrrichtlinie -- Persistenz des Domänenstatus (SQLite-Durchschreibcache für Fallbacks, Budgets, Sperrungen, Leistungsschalter) -- Richtlinien-Engine für zentralisierte Anfrageauswertung (Sperrung → Budget → Fallback) -- Fordern Sie Telemetrie mit p50/p95/p99-Latenzaggregation an -- Korrelations-ID (X-Request-Id) für eine durchgängige Nachverfolgung -- Compliance-Audit-Protokollierung mit Opt-out pro API-Schlüssel -- Evaluierungsrahmen für die LLM-Qualitätssicherung -- Resilience-UI-Dashboard mit Echtzeit-Leistungsschalterstatus -- Modulare OAuth-Anbieter (12 einzelne Module unter „src/lib/oauth/providers/“) +## Executive Summary -Primäres Laufzeitmodell: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -– Next.js-App-Routen unter „src/app/api/_“ implementieren sowohl Dashboard-APIs als auch Kompatibilitäts-APIs -– Ein gemeinsam genutzter SSE/Routing-Kern in „src/sse/_“ + „open-sse/\*“ kümmert sich um die Ausführung, Übersetzung, Streaming, Fallback und Nutzung des Anbieters## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Lokale Gateway-Laufzeit -- Dashboard-Verwaltungs-APIs -- Anbieterauthentifizierung und Token-Aktualisierung -- Fordern Sie Übersetzung und SSE-Streaming an -- Lokaler Status + Nutzungspersistenz -- Optionale Orchestrierung der Cloud-Synchronisierung### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Cloud-Service-Implementierung hinter „NEXT_PUBLIC_CLOUD_URL“. -- Anbieter-SLA/Kontrollebene außerhalb des lokalen Prozesses -- Externe CLI-Binärdateien selbst (Claude CLI, Codex CLI usw.)## Dashboard Surface (Current) +### Out of Scope -Hauptseiten unter „src/app/(dashboard)/dashboard/“: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- „/dashboard“ – Schnellstart + Anbieterübersicht -- „/dashboard/endpoint“ – Endpunkt-Proxy + MCP + A2A + API-Endpunkt-Registerkarten -- „/dashboard/providers“ – Anbieterverbindungen und Anmeldeinformationen -- „/dashboard/combos“ – Kombinationsstrategien, Vorlagen, Modell-Routing-Regeln -- „/dashboard/costs“ – Kostenaggregation und Preistransparenz -- „/dashboard/analytics“ – Nutzungsanalysen und Auswertungen -- „/dashboard/limits“ – Kontingent-/Ratenkontrolle -- „/dashboard/cli-tools“ – CLI-Onboarding, Laufzeiterkennung, Konfigurationsgenerierung -- „/dashboard/agents“ – erkannte ACP-Agenten + benutzerdefinierte Agentenregistrierung -- „/dashboard/media“ – Bild-/Video-/Musikspielplatz -- „/dashboard/search-tools“ – Tests und Verlauf des Suchanbieters -- „/dashboard/health“ – Betriebszeit, Leistungsschalter, Ratenbegrenzungen -- „/dashboard/logs“ – Anforderungs-/Proxy-/Audit-/Konsolenprotokolle -- „/dashboard/settings“ – Registerkarten für Systemeinstellungen (Allgemein, Routing, Combo-Standardeinstellungen usw.) -- „/dashboard/api-manager“ – API-Schlüssellebenszyklus und Modellberechtigungen## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,140 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Hauptverzeichnisse: +Main directories: -- „src/app/api/v1/_“ und „src/app/api/v1beta/_“ für Kompatibilitäts-APIs -- „src/app/api/\*“ für Verwaltungs-/Konfigurations-APIs -- Next schreibt in „next.config.mjs“ die Zuordnung von „/v1/_“ zu „/api/v1/_“ um +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Wichtige Kompatibilitätsrouten: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- „src/app/api/v1/models/route.ts“ – enthält benutzerdefinierte Modelle mit „custom: true“. -- „src/app/api/v1/embeddings/route.ts“ – Einbettungsgenerierung (6 Anbieter) -- `src/app/api/v1/images/generations/route.ts` — Bildgenerierung (4+ Anbieter inkl. Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- „src/app/api/v1/providers/[provider]/chat/completions/route.ts“ – dedizierter Chat pro Anbieter -- „src/app/api/v1/providers/[provider]/embeddings/route.ts“ – dedizierte Einbettungen pro Anbieter -- „src/app/api/v1/providers/[provider]/images/generations/route.ts“ – dedizierte Bilder pro Anbieter +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` -- `src/app/api/v1beta/models/[...pfad]/route.ts` +- `src/app/api/v1beta/models/[...path]/route.ts` -Verwaltungsdomänen: +Management domains: -- Authentifizierung/Einstellungen: `src/app/api/auth/*`, `src/app/api/settings/*` -- Anbieter/Verbindungen: `src/app/api/providers*` -- Anbieterknoten: `src/app/api/provider-nodes*` -- Benutzerdefinierte Modelle: `src/app/api/provider-models` (GET/POST/DELETE) -- Modellkatalog: `src/app/api/models/route.ts` (GET) -- Proxy-Konfiguration: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- Schlüssel/Aliase/Combos/Preise: „src/app/api/keys*“, „src/app/api/models/alias“, „src/app/api/combos*“, „src/app/api/pricing“. -- Verwendung: `src/app/api/usage/*` -- Synchronisierung/Cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` -- CLI-Tool-Helfer: `src/app/api/cli-tools/*` -- IP-Filter: `src/app/api/settings/ip-filter` (GET/PUT) -- Thinking-Budget: `src/app/api/settings/thinking-budget` (GET/PUT) -- Systemeingabeaufforderung: `src/app/api/settings/system-prompt` (GET/PUT) -- Sitzungen: `src/app/api/sessions` (GET) -- Ratenlimits: `src/app/api/rate-limits` (GET) - – Resilienz: „src/app/api/resilience“ (GET/PATCH) – Anbieterprofile, Leistungsschalter, Ratengrenzstatus -- Resilience-Reset: `src/app/api/resilience/reset` (POST) – Breaker + Abklingzeiten zurücksetzen -- Cache-Statistiken: `src/app/api/cache/stats` (GET/DELETE) -- Modellverfügbarkeit: `src/app/api/models/availability` (GET/POST) -- Telemetrie: `src/app/api/telemetry/summary` (GET) +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) - Budget: `src/app/api/usage/budget` (GET/POST) -- Fallback-Ketten: `src/app/api/fallback/chains` (GET/POST/DELETE) -- Compliance-Audit: `src/app/api/compliance/audit-log` (GET) -- Auswertungen: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- Richtlinien: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -Hauptflussmodule: +## 2) SSE + Translation Core -- Eintrag: `src/sse/handlers/chat.ts` -- Kernorchestrierung: „open-sse/handlers/chatCore.ts“. -- Anbieterausführungsadapter: „open-sse/executors/\*“. -- Formaterkennung/Anbieterkonfiguration: „open-sse/services/provider.ts“. -- Modellanalyse/-auflösung: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Konto-Fallback-Logik: „open-sse/services/accountFallback.ts“. -- Übersetzungsregister: „open-sse/translator/index.ts“. -- Stream-Transformationen: „open-sse/utils/stream.ts“, „open-sse/utils/streamHandler.ts“. -- Nutzungsextraktion/-normalisierung: `open-sse/utils/usageTracking.ts` -- Think-Tag-Parser: „open-sse/utils/thinkTagParser.ts“. -- Einbettungshandler: `open-sse/handlers/embeddings.ts` -- Einbettungsanbieter-Registrierung: „open-sse/config/embeddingRegistry.ts“. -- Handler für die Bildgenerierung: „open-sse/handlers/imageGeneration.ts“. -- Registrierung des Bildanbieters: „open-sse/config/imageRegistry.ts“. -- Antwortbereinigung: `open-sse/handlers/responseSanitizer.ts` -- Rollennormalisierung: `open-sse/services/roleNormalizer.ts` +Main flow modules: -Dienste (Geschäftslogik): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- Kontoauswahl/-bewertung: `open-sse/services/accountSelector.ts` -- Kontextlebenszyklusverwaltung: „open-sse/services/contextManager.ts“. -- Durchsetzung des IP-Filters: „open-sse/services/ipFilter.ts“. -- Sitzungsverfolgung: `open-sse/services/sessionManager.ts` -- Deduplizierung anfordern: „open-sse/services/signatureCache.ts“. -- System-Prompt-Injection: „open-sse/services/systemPrompt.ts“. -- Thinking Budget Management: „open-sse/services/thinkingBudget.ts“. -- Wildcard-Modell-Routing: „open-sse/services/wildcardRouter.ts“. -- Ratenlimitverwaltung: `open-sse/services/rateLimitManager.ts` -- Leistungsschalter: `open-sse/services/CircuitBreaker.ts` +Services (business logic): -Module der Domänenschicht: +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- Modellverfügbarkeit: `src/lib/domain/modelAvailability.ts` -- Kostenregeln/Budgets: `src/lib/domain/costRules.ts` -- Fallback-Richtlinie: `src/lib/domain/fallbackPolicy.ts` -- Combo-Resolver: `src/lib/domain/comboResolver.ts` -- Sperrrichtlinie: `src/lib/domain/lockoutPolicy.ts` - – Richtlinien-Engine: „src/domain/policyEngine.ts“ – zentralisierte Sperrung → Budget → Fallback-Auswertung -- Fehlercodekatalog: `src/lib/domain/errorCodes.ts` -- Anforderungs-ID: `src/lib/domain/requestId.ts` -- Abrufzeitüberschreitung: `src/lib/domain/fetchTimeout.ts` -- Telemetrie anfordern: `src/lib/domain/requestTelemetry.ts` -- Compliance/Audit: `src/lib/domain/compliance/index.ts` -- Eval-Runner: `src/lib/domain/evalRunner.ts` - – Domänenstatus-Persistenz: „src/lib/db/domainState.ts“ – SQLite CRUD für Fallback-Ketten, Budgets, Kostenverlauf, Sperrstatus, Leistungsschalter +Domain layer modules: -OAuth-Provider-Module (12 einzelne Dateien unter „src/lib/oauth/providers/“): +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -- Registrierungsindex: `src/lib/oauth/providers/index.ts` -- Einzelne Anbieter: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` - – Thin Wrapper: „src/lib/oauth/providers.ts“ – erneuter Export aus einzelnen Modulen## 3) Persistence Layer +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -Primärstatus-DB (SQLite): +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -- Kerninfra: `src/lib/db/core.ts` (better-sqlite3, Migrationen, WAL) -- Fassade erneut exportieren: `src/lib/localDb.ts` (dünne Kompatibilitätsschicht für Aufrufer) -- Datei: „${DATA_DIR}/storage.sqlite“ (oder „$XDG_CONFIG_HOME/omniroute/storage.sqlite“, wenn festgelegt, sonst „~/.omniroute/storage.sqlite“) -- Entitäten (Tabellen + KV-Namespaces): ProviderConnections, ProviderNodes, ModelAliases, Combos, APIKeys, Einstellungen, Preise,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +## 3) Persistence Layer -Nutzungsdauer: +Primary state DB (SQLite): -- Fassade: `src/lib/usageDb.ts` (zerlegte Module in `src/lib/usage/*`) -- SQLite-Tabellen in „storage.sqlite“: „usage_history“, „call_logs“, „proxy_logs“. -- Optionale Dateiartefakte bleiben aus Kompatibilitäts-/Debuggründen erhalten (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) - – Ältere JSON-Dateien werden durch Startmigrationen nach SQLite migriert, sofern vorhanden +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** + +Usage persistence: + +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present Domain State DB (SQLite): -– „src/lib/db/domainState.ts“ – CRUD-Operationen für den Domänenstatus +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start -- Tabellen (erstellt in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_Circuit_breakers` -- Write-Through-Cache-Muster: In-Memory-Maps sind zur Laufzeit maßgeblich; Mutationen werden synchron zu SQLite geschrieben; Der Status wird beim Kaltstart aus der DB wiederhergestellt## 4) Auth + Security Surfaces +## 4) Auth + Security Surfaces -- Dashboard-Cookie-Authentifizierung: „src/proxy.ts“, „src/app/api/auth/login/route.ts“. -- API-Schlüsselgenerierung/-überprüfung: `src/shared/utils/apiKey.ts` - – Provider-Geheimnisse blieben in „providerConnections“-Einträgen bestehen -- Unterstützung für ausgehende Proxys über „open-sse/utils/proxyFetch.ts“ (env vars) und „open-sse/utils/networkProxy.ts“ (pro Anbieter oder global konfigurierbar)## 5) Cloud Sync +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) -- Scheduler-Init: „src/lib/initCloudSync.ts“, „src/shared/services/initializeCloudSync.ts“, „src/shared/services/modelSyncScheduler.ts“. -- Periodische Aufgabe: `src/shared/services/cloudSyncScheduler.ts` -- Periodische Aufgabe: `src/shared/services/modelSyncScheduler.ts` -- Kontrollroute: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -339,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Fallback-Entscheidungen werden von „open-sse/services/accountFallback.ts“ unter Verwendung von Statuscodes und Fehlermeldungsheuristiken gesteuert. Combo-Routing fügt einen zusätzlichen Schutz hinzu: 400-Fehler im Anbieterbereich wie Upstream-Inhaltsblockierungs- und Rollenvalidierungsfehler werden als modelllokale Fehler behandelt, sodass spätere Combo-Ziele weiterhin ausgeführt werden können.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -369,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -Die Aktualisierung während des Live-Verkehrs wird in „open-sse/handlers/chatCore.ts“ über den Executor „refreshCredentials()“ ausgeführt.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -401,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -Die regelmäßige Synchronisierung wird durch „CloudSyncScheduler“ ausgelöst, wenn die Cloud aktiviert ist.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -502,12 +532,14 @@ erDiagram } ``` -Physische Speicherdateien: +Physical storage files: -- Primäre Laufzeit-DB: „${DATA_DIR}/storage.sqlite“. -- Protokollzeilen anfordern: „${DATA_DIR}/log.txt“ (Kompatibilitäts-/Debug-Artefakt) -- Strukturierte Anrufnutzlastarchive: „${DATA_DIR}/call_logs/“. -- optionale Übersetzer-/Request-Debug-Sitzungen: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -542,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: Kompatibilitäts-APIs -- `src/app/api/v1/providers/[provider]/*`: dedizierte Routen pro Anbieter (Chat, Einbettungen, Bilder) -- „src/app/api/providers\*“: Anbieter-CRUD, Validierung, Tests -- „src/app/api/provider-nodes\*“: benutzerdefinierte kompatible Knotenverwaltung -- „src/app/api/provider-models“: benutzerdefinierte Modellverwaltung (CRUD) -- `src/app/api/models/route.ts`: Modellkatalog-API (Aliase + benutzerdefinierte Modelle) -- `src/app/api/oauth/*`: OAuth/Gerätecodeflüsse -- „src/app/api/keys\*“: Lebenszyklus des lokalen API-Schlüssels -- `src/app/api/models/alias`: Alias-Verwaltung -- `src/app/api/combos*`: Fallback-Combo-Verwaltung -- „src/app/api/pricing“: Preisüberschreibungen für die Kostenberechnung -- `src/app/api/settings/proxy`: Proxy-Konfiguration (GET/PUT/DELETE) -- „src/app/api/settings/proxy/test“: Test der ausgehenden Proxy-Konnektivität (POST) -- `src/app/api/usage/*`: Nutzungs- und Protokoll-APIs -- `src/app/api/sync/*` + `src/app/api/cloud/*`: Cloud-Synchronisierung und Cloud-orientierte Helfer -- `src/app/api/cli-tools/*`: lokale CLI-Konfigurationsschreiber/-prüfer -- `src/app/api/settings/ip-filter`: IP-Zulassungsliste/Blockliste (GET/PUT) -- `src/app/api/settings/thinking-budget`: Thinking-Token-Budgetkonfiguration (GET/PUT) -- `src/app/api/settings/system-prompt`: globale Systemeingabeaufforderung (GET/PUT) -- `src/app/api/sessions`: Auflistung der aktiven Sitzungen (GET) -- `src/app/api/rate-limits`: Status des Ratenlimits pro Konto (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- „src/sse/handlers/chat.ts“: Anforderungsanalyse, Kombinationsverarbeitung, Kontoauswahlschleife -- „open-sse/handlers/chatCore.ts“: Übersetzung, Executor-Dispatch, Wiederholungs-/Aktualisierungsbehandlung, Stream-Setup -- „open-sse/executors/\*“: anbieterspezifisches Netzwerk- und Formatverhalten### Translation Registry and Format Converters +### Routing and Execution Core -- „open-sse/translator/index.ts“: Übersetzerregistrierung und Orchestrierung -- Übersetzer anfordern: `open-sse/translator/request/*` -- Antwortübersetzer: `open-sse/translator/response/*` -- Formatkonstanten: „open-sse/translator/formats.ts“.### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: persistente Konfiguration/Status und Domänenpersistenz auf SQLite -- `src/lib/localDb.ts`: Kompatibilitäts-Neuexport für DB-Module -- „src/lib/usageDb.ts“: Fassade der Nutzungshistorie/Anrufprotokolle über SQLite-Tabellen## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Jeder Anbieter verfügt über einen speziellen Executor, der „BaseExecutor“ (in „open-sse/executors/base.ts“) erweitert und URL-Erstellung, Header-Konstruktion, Wiederholung mit exponentiellem Backoff, Hooks für die Aktualisierung von Anmeldeinformationen und die Orchestrierungsmethode „execute()“ bereitstellt. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Testamentsvollstrecker | Anbieter(n) | Besondere Handhabung | -| ---------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------------------------------------------- | -| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamische URL-/Header-Konfiguration pro Anbieter | -| `AntigravityExecutor` | Google Antigravitation | Benutzerdefinierte Projekt-/Sitzungs-IDs, Wiederholen nach dem Parsen | -| `CodexExecutor` | OpenAI-Codex | Fügt Systemanweisungen ein und erzwingt den Denkaufwand | -| `CursorExecutor` | Cursor-IDE | ConnectRPC-Protokoll, Protobuf-Kodierung, Anforderungssignatur über Prüfsumme | -| `GithubExecutor` | GitHub-Copilot | Copilot-Token-Aktualisierung, VSCode-imitierende Header | -| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream-Binärformat → SSE-Konvertierung | -| `GeminiCLIExecutor` | Gemini CLI | Aktualisierungszyklus des Google OAuth-Tokens | +### Persistence -Alle anderen Anbieter (einschließlich benutzerdefinierter kompatibler Knoten) verwenden den „DefaultExecutor“.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Anbieter | Formatieren | Authentifizierung | Stream | Nicht-Stream | Token-Aktualisierung | Nutzungs-API | -| ---------------- | ---------------- | ---------------------------- | ---------------- | ------------ | -------------------- | ------------------------------ | ------------------------------ | -| Claude | Claude | API-Schlüssel / OAuth | ✅ | ✅ | ✅ | ⚠️ Nur Administrator | -| Zwillinge | Zwillinge | API-Schlüssel / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud-Konsole | -| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud-Konsole | -| Antigravitation | Antigravitation | OAuth | ✅ | ✅ | ✅ | ✅ Vollständige Kontingent-API | -| OpenAI | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ | -| Kodex | Openai-Antworten | OAuth | ✅ gezwungen | ❌ | ✅ | ✅ Tariflimits | -| GitHub-Copilot | openai | OAuth + Copilot-Token | ✅ | ✅ | ✅ | ✅ Kontingent-Snapshots | -| Cursor | Cursor | Benutzerdefinierte Prüfsumme | ✅ | ✅ | ❌ | ❌ | -| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Nutzungsbeschränkungen | -| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Auf Anfrage | -| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Auf Anfrage | -| OpenRouter | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | Claude | API-Schlüssel | ✅ | ✅ | ❌ | ❌ | -| DeepSeek | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ | -| Groq | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ | -| Mistral | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ | -| Ratlosigkeit | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ | -| Zusammen KI | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ | -| Feuerwerk KI | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ | -| Großhirn | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ | -| Kohärent | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Zu den erkannten Quellformaten gehören: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. + +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | + +All other providers (including custom compatible nodes) use the `DefaultExecutor`. + +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: - `openai` -- „Openai-Antworten“. -- `Claude` -- „Zwillinge“. +- `openai-responses` +- `claude` +- `gemini` -Zu den Zielformaten gehören: +Target formats include: -- OpenAI-Chat/Antworten +- OpenAI chat/Responses - Claude -- Gemini/Gemini-CLI/Antigravity-Umschlag +- Gemini/Gemini-CLI/Antigravity envelope - Kiro - Cursor -Übersetzungen verwenden**OpenAI als Hub-Format**– alle Konvertierungen durchlaufen OpenAI als Zwischenformat:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Übersetzungen werden dynamisch basierend auf der Form der Quellnutzlast und dem Zielformat des Anbieters ausgewählt. +Additional processing layers in the translation pipeline: -Zusätzliche Verarbeitungsebenen in der Übersetzungspipeline: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Antwortbereinigung**– Entfernt nicht standardmäßige Felder aus Antworten im OpenAI-Format (sowohl Streaming als auch Nicht-Streaming), um eine strikte SDK-Konformität sicherzustellen --**Rollennormalisierung**– Konvertiert „Entwickler“ → „System“ für Nicht-OpenAI-Ziele; führt „System“ → „Benutzer“ für Modelle zusammen, die die Systemrolle ablehnen (GLM, ERNIE) --**Think-Tag-Extraktion**– Analysiert „...“-Blöcke aus dem Inhalt in das Feld „reasoning_content“. --**Strukturierte Ausgabe**– Konvertiert OpenAI „response_format.json_schema“ in „responseMimeType“ + „responseSchema“ von Gemini## Supported API Endpoints +## Supported API Endpoints -| Endpunkt | Formatieren | Handler | -| ------------------------------------------------- | ------------------- | ------------------------------------------------------------------- | -| `POST /v1/chat/completions` | OpenAI-Chat | `src/sse/handlers/chat.ts` | -| `POST /v1/messages` | Claude-Nachrichten | Gleicher Handler (automatisch erkannt) | -| `POST /v1/responses` | OpenAI-Antworten | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/embeddings` | OpenAI-Einbettungen | `open-sse/handlers/embeddings.ts` | -| `GET /v1/embeddings` | Modellliste | API-Route | -| `POST /v1/images/generations` | OpenAI-Bilder | `open-sse/handlers/imageGeneration.ts` | -| `GET /v1/images/generations` | Modellliste | API-Route | -| `POST /v1/providers/{provider}/chat/completions` | OpenAI-Chat | Dedizierter pro Anbieter mit Modellvalidierung | -| `POST /v1/providers/{provider}/embeddings` | OpenAI-Einbettungen | Dedizierter pro Anbieter mit Modellvalidierung | -| `POST /v1/providers/{provider}/images/generations` | OpenAI-Bilder | Dedizierter pro Anbieter mit Modellvalidierung | -| `POST /v1/messages/count_tokens` | Claude Token Count | API-Route | -| `GET /v1/models` | Liste der OpenAI-Modelle | API-Route (Chat + Einbettung + Bild + benutzerdefinierte Modelle) | -| `GET /api/models/catalog` | Katalog | Alle Modelle gruppiert nach Anbieter + Typ | -| `POST /v1beta/models/*:streamGenerateContent` | Zwillinge heimisch | API-Route | -| `GET/PUT/DELETE /api/settings/proxy` | Proxy-Konfiguration | Netzwerk-Proxy-Konfiguration | -| `POST /api/settings/proxy/test` | Proxy-Konnektivität | Proxy-Zustands-/Konnektivitätstest-Endpunkt | -| `GET/POST/DELETE /api/provider-models` | Anbietermodelle | Metadaten des Anbietermodells, die benutzerdefinierte und verwaltete verfügbare Modelle unterstützen |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -Der Bypass-Handler („open-sse/utils/bypassHandler.ts“) fängt bekannte „Wegwerf“-Anfragen von Claude CLI ab – Warmup-Pings, Titelextraktionen und Token-Zählungen – und gibt eine**falsche Antwort**zurück, ohne Upstream-Anbieter-Tokens zu verbrauchen. Dies wird nur ausgelöst, wenn „User-Agent“ „claude-cli“ enthält.## Request Logger Pipeline +## Bypass Handler -Der Anforderungslogger („open-sse/utils/requestLogger.ts“) bietet eine 7-stufige Debug-Protokollierungspipeline, die standardmäßig deaktiviert und über „ENABLE_REQUEST_LOGS=true“ aktiviert ist:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Dateien werden für jede Anforderungssitzung in „/logs//“ geschrieben.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- Abklingzeit des Anbieterkontos bei vorübergehenden/Raten-/Authentifizierungsfehlern -- Konto-Fallback vor fehlgeschlagener Anfrage -- Combo-Modell-Fallback, wenn der aktuelle Modell-/Anbieterpfad erschöpft ist## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- Vorabprüfung und Aktualisierung mit erneutem Versuch für aktualisierbare Anbieter - – 401/403-Wiederholungsversuch nach Aktualisierungsversuch im Kernpfad## 3) Stream Safety +## 2) Token Expiry -- Trennungsfähiger Stream-Controller -- Übersetzungsstream mit Stream-Ende-Flush und „[FERTIG]“-Behandlung -- Fallback der Nutzungsschätzung, wenn Metadaten zur Anbieternutzung fehlen## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -– Synchronisierungsfehler werden angezeigt, die lokale Laufzeit wird jedoch fortgesetzt -– Der Scheduler verfügt über eine wiederholfähige Logik, aber die regelmäßige Ausführung ruft derzeit standardmäßig eine Einzelversuchssynchronisierung auf## 5) Data Integrity +## 3) Stream Safety -- SQLite-Schemamigrationen und automatische Upgrade-Hooks beim Start -- Legacy-JSON → SQLite-Migrationskompatibilitätspfad## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Quellen für die Laufzeitsichtbarkeit: +## 4) Cloud Sync Degradation -- Konsolenprotokolle von „src/sse/utils/logger.ts“. -- Nutzungsaggregate pro Anfrage in SQLite („usage_history“, „call_logs“, „proxy_logs“) -- Vierstufige detaillierte Nutzlasterfassungen in SQLite (`request_detail_logs`), wenn `settings.detailed_logs_enabled=true` -- Statusprotokoll der Textanfrage in „log.txt“ (optional/kompatibel) -- optionale tiefe Anforderungs-/Übersetzungsprotokolle unter „logs/“, wenn „ENABLE_REQUEST_LOGS=true“ ist -- Dashboard-Nutzungsendpunkte (`/api/usage/*`) für die UI-Nutzung +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -Die detaillierte Anforderungsnutzlasterfassung speichert bis zu vier JSON-Nutzlaststufen pro weitergeleitetem Anruf: +## 5) Data Integrity -- Rohanfrage vom Client erhalten -- Die übersetzte Anfrage wurde tatsächlich an den Upstream gesendet -- Anbieterantwort als JSON rekonstruiert; Gestreamte Antworten werden zur endgültigen Zusammenfassung plus Stream-Metadaten komprimiert - – endgültige Client-Antwort, die von OmniRoute zurückgegeben wird; Gestreamte Antworten werden in derselben kompakten Zusammenfassungsform gespeichert## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- JWT-Geheimnis („JWT_SECRET“) sichert die Überprüfung/Signatur von Dashboard-Sitzungscookies - – Der anfängliche Passwort-Bootstrap („INITIAL_PASSWORD“) sollte explizit für die erstmalige Bereitstellung konfiguriert werden -- Das HMAC-Geheimnis des API-Schlüssels („API_KEY_SECRET“) sichert das generierte lokale API-Schlüsselformat - – Anbietergeheimnisse (API-Schlüssel/Tokens) werden in der lokalen Datenbank gespeichert und sollten auf Dateisystemebene geschützt werden - – Cloud-Synchronisierungsendpunkte basieren auf der API-Schlüsselauthentifizierung und der Maschinen-ID-Semantik## Environment and Runtime Matrix +## Observability and Operational Signals -Vom Code aktiv verwendete Umgebungsvariablen: +Runtime visibility sources: -- App/Auth: „JWT_SECRET“, „INITIAL_PASSWORD“. -- Speicher: `DATA_DIR` -- Kompatibles Knotenverhalten: „ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE“. -- Optionale Speicherbasisüberschreibung (Linux/macOS, wenn „DATA_DIR“ nicht festgelegt ist): „XDG_CONFIG_HOME“. -- Sicherheits-Hashing: „API_KEY_SECRET“, „MACHINE_ID_SALT“. -- Protokollierung: „ENABLE_REQUEST_LOGS“. -- Synchronisierungs-/Cloud-URLing: „NEXT_PUBLIC_BASE_URL“, „NEXT_PUBLIC_CLOUD_URL“. -- Ausgehender Proxy: „HTTP_PROXY“, „HTTPS_PROXY“, „ALL_PROXY“, „NO_PROXY“ und Varianten in Kleinbuchstaben -- SOCKS5-Funktionsflags: „ENABLE_SOCKS5_PROXY“, „NEXT_PUBLIC_ENABLE_SOCKS5_PROXY“. -- Plattform-/Laufzeit-Helfer (keine App-spezifische Konfiguration): „APPDATA“, „NODE_ENV“, „PORT“, „HOSTNAME“.## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. „usageDb“ und „localDb“ verwenden dieselbe Basisverzeichnisrichtlinie („DATA_DIR“ -> „XDG_CONFIG_HOME/omniroute“ -> „~/.omniroute“) bei der Migration älterer Dateien. -2. „/api/v1/route.ts“ delegiert an denselben einheitlichen Katalog-Builder, der von „/api/v1/models“ verwendet wird (`src/app/api/v1/models/catalog.ts`), um semantische Abweichungen zu vermeiden. -3. Der Anforderungslogger schreibt bei Aktivierung vollständige Header/Textkörper. Behandeln Sie das Protokollverzeichnis als vertraulich. -4. Das Cloud-Verhalten hängt von der korrekten „NEXT_PUBLIC_BASE_URL“ und der Erreichbarkeit des Cloud-Endpunkts ab. -5. Das Verzeichnis „open-sse/“ wird als „@omniroute/open-sse“**npm-Workspace-Paket**veröffentlicht. Der Quellcode importiert es über „@omniroute/open-sse/...“ (aufgelöst durch Next.js „transpilePackages“). Dateipfade in diesem Dokument verwenden aus Konsistenzgründen weiterhin den Verzeichnisnamen „open-sse/“. -6. Diagramme im Dashboard verwenden**Recharts**(SVG-basiert) für zugängliche, interaktive Analysevisualisierungen (Modellnutzungs-Balkendiagramme, Anbieteraufschlüsselungstabellen mit Erfolgsquoten). -7. E2E-Tests verwenden**Playwright**(`tests/e2e/`) und werden über `npm run test:e2e` ausgeführt. Unit-Tests verwenden**Node.js Test Runner**(`tests/unit/`) und werden über `npm run test:unit` ausgeführt. Der Quellcode unter „src/“ ist**TypeScript**(`.ts`/`.tsx`); Der `open-sse/`-Arbeitsbereich bleibt JavaScript (`.js`). -8. Die Einstellungsseite ist in 5 Registerkarten unterteilt: Sicherheit, Routing (6 globale Strategien: Fill-First, Round-Robin, P2C, Random, Least-Used, Cost-Optimized), Resilience (bearbeitbare Ratenlimits, Leistungsschalter, Richtlinien), AI (Thinking Budget, System Prompt, Prompt Cache), Advanced (Proxy).## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- Aus der Quelle erstellen: „npm run build“. -- Docker-Image erstellen: `docker build -t omniroute .` -- Starten Sie den Dienst und überprüfen Sie: +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: - `GET /api/settings` - `GET /api/v1/models` - – Die Basis-URL des CLI-Ziels sollte „http://:20128/v1“ lauten, wenn „PORT=20128“. +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/de/docs/FEATURES.md b/docs/i18n/de/docs/FEATURES.md index 926bd473db..d76662f8aa 100644 --- a/docs/i18n/de/docs/FEATURES.md +++ b/docs/i18n/de/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Visuelle Anleitung zu jedem Abschnitt des OmniRoute-Dashboards.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Verwalten Sie KI-Anbieterverbindungen: OAuth-Anbieter (Claude Code, Codex, Gemini CLI), API-Schlüsselanbieter (Groq, DeepSeek, OpenRouter) und kostenlose Anbieter (Qoder, Qwen, Kiro). Bei Kiro-Konten ist die Nachverfolgung des Guthabens möglich – verbleibende Guthaben, Gesamtguthaben und Verlängerungsdatum sind im Dashboard → Nutzung sichtbar.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Erstellen Sie Modell-Routing-Kombinationen mit 6 Strategien: Priorität, gewichtet, Round-Robin, zufällig, am wenigsten verwendet und kostenoptimiert. Jede Kombination verkettet mehrere Modelle mit automatischem Fallback und umfasst schnelle Vorlagen und Bereitschaftsprüfungen.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Umfassende Nutzungsanalysen mit Token-Verbrauch, Kostenschätzungen, Aktivitäts-Heatmaps, wöchentlichen Verteilungsdiagrammen und Aufschlüsselungen pro Anbieter.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Echtzeitüberwachung: Betriebszeit, Speicher, Version, Latenzperzentile (p50/p95/p99), Cache-Statistiken und Leistungsschalterzustände des Anbieters.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Vier Modi zum Debuggen von API-Übersetzungen:**Playground**(Formatkonverter),**Chat Tester**(Live-Anfragen),**Test Bench**(Batch-Tests) und**Live Monitor**(Echtzeit-Stream).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Testen Sie jedes Modell direkt vom Dashboard aus. Wählen Sie Anbieter, Modell und Endpunkt aus, schreiben Sie Eingabeaufforderungen mit Monaco Editor, streamen Sie Antworten in Echtzeit, brechen Sie mitten im Stream ab und sehen Sie sich Timing-Metriken an.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Anpassbare Farbthemen für das gesamte Dashboard. Wählen Sie aus 7 voreingestellten Farben (Koralle, Blau, Rot, Grün, Violett, Orange, Cyan) oder erstellen Sie ein individuelles Design, indem Sie eine beliebige Hex-Farbe auswählen. Unterstützt Hell-, Dunkel- und Systemmodus.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Umfangreiches Einstellungsfeld mit Registerkarten: +Comprehensive settings panel with tabs: --**Allgemein**– Systemspeicher, Backup-Management (Datenbank exportieren/importieren) -**Erscheinungsbild**– Themenauswahl (Dunkel/Hell/System), Voreinstellungen für Farbthemen und benutzerdefinierte Farben, Sichtbarkeit des Gesundheitsprotokolls, Steuerelemente für die Sichtbarkeit von Elementen in der Seitenleiste -**Sicherheit**– API-Endpunktschutz, benutzerdefinierte Anbieterblockierung, IP-Filterung, Sitzungsinformationen -**Routing**– Modellaliase, Verschlechterung der Hintergrundaufgabe -**Resilienz**– Persistenz der Ratenbegrenzung, Leistungsschalter-Optimierung, automatische Deaktivierung gesperrter Konten, Überwachung des Anbieterablaufs -**Erweitert**– Konfigurationsüberschreibungen, Konfigurations-Audit-Trail, Fallback-Verschlechterungsmodus![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Ein-Klick-Konfiguration für KI-Codierungstools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor und Factory Droid. Bietet automatisches Anwenden/Zurücksetzen der Konfiguration, Verbindungsprofile und Modellzuordnung.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Dashboard zum Erkennen und Verwalten von CLI-Agenten. Zeigt ein Raster mit 14 integrierten Agenten (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) mit: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Installationsstatus**– Installiert/Nicht gefunden mit Versionserkennung -**Protokollabzeichen**– stdio, HTTP usw. -**Benutzerdefinierte Agents**– Registrieren Sie jedes CLI-Tool über ein Formular (Name, Binärdatei, Versionsbefehl, Spawn-Argumente). -**CLI-Fingerabdruck-Abgleich**– Umschalten pro Anbieter, um native CLI-Anfragesignaturen abzugleichen, wodurch das Verbotsrisiko verringert und gleichzeitig die Proxy-IP erhalten bleibt--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Generieren Sie Bilder, Videos und Musik über das Dashboard. Unterstützt OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open und MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Echtzeit-Anfrageprotokollierung mit Filterung nach Anbieter, Modell, Konto und API-Schlüssel. Zeigt Statuscodes, Token-Nutzung, Latenz und Antwortdetails an.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Ihr einheitlicher API-Endpunkt mit Aufschlüsselung der Funktionen: Chat-Abschlüsse, Antwort-API, Einbettungen, Bildgenerierung, Neuranking, Audiotranskription, Text-to-Speech, Moderationen und registrierte API-Schlüssel. Cloudflare Quick Tunnel-Integration und Cloud-Proxy-Unterstützung für Fernzugriff.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -API-Schlüssel erstellen, festlegen und widerrufen. Jeder Schlüssel kann auf bestimmte Modelle/Anbieter mit Vollzugriff oder Nur-Lese-Berechtigungen beschränkt werden. Visuelle Schlüsselverwaltung mit Nutzungsverfolgung.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Verwaltungsaktionsverfolgung mit Filterung nach Aktionstyp, Akteur, Ziel, IP-Adresse und Zeitstempel. Vollständiger Sicherheitsereignisverlauf.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Native Electron-Desktop-App für Windows, macOS und Linux. Führen Sie OmniRoute als eigenständige Anwendung mit Taskleistenintegration, Offline-Unterstützung, automatischer Aktualisierung und Installation mit einem Klick aus. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Hauptmerkmale: +Key features: -- Abfrage der Serverbereitschaft (kein leerer Bildschirm beim Kaltstart) -- Taskleiste mit Portverwaltung -- Inhaltssicherheitsrichtlinie -- Einzelinstanzsperre -- Automatische Aktualisierung beim Neustart -- Plattformabhängige Benutzeroberfläche (Ampeln für macOS, Standardtitelleiste für Windows/Linux) -- Hardened Electron Build-Paketierung – symbolisch verknüpfte „node_modules“ im Standalone-Bundle werden vor dem Paketieren erkannt und abgelehnt, wodurch eine Laufzeitabhängigkeit von der Build-Maschine verhindert wird (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Die vollständige Dokumentation finden Sie unter [`electron/README.md`](../electron/README.md). +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/de/docs/TROUBLESHOOTING.md b/docs/i18n/de/docs/TROUBLESHOOTING.md index c1dcb822a2..8dfa2dffff 100644 --- a/docs/i18n/de/docs/TROUBLESHOOTING.md +++ b/docs/i18n/de/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -Häufige Probleme und Lösungen für OmniRoute.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Problem | Lösung | -| ------------------------------------------ | ------------------------------------------------------------------------------------- | --- | -| Erster Login funktioniert nicht | Legen Sie „INITIAL_PASSWORD“ in „.env“ fest (keine fest codierte Standardeinstellung) | -| Dashboard wird am falschen Port geöffnet | Setzen Sie „PORT=20128“ und „NEXT_PUBLIC_BASE_URL=http://localhost:20128“ | -| Keine Anforderungsprotokolle unter „logs/“ | Setzen Sie „ENABLE_REQUEST_LOGS=true“ | -| EACCES: Berechtigung verweigert | Setzen Sie „DATA_DIR=/path/to/writable/dir“, um „~/.omniroute“ zu überschreiben | -| Routing-Strategie wird nicht gespeichert | Update auf v1.4.11+ (Zod-Schema-Korrektur für Einstellungspersistenz) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Ursache:**Anbieterkontingent erschöpft. +**Cause:** Provider quota exhausted. **Fix:** -1. Überprüfen Sie den Quoten-Tracker im Dashboard -2. Verwenden Sie eine Kombination mit Fallback-Stufen -3. Wechseln Sie zum günstigeren/kostenlosen Tarif### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Ursache:**Das Abonnementkontingent ist erschöpft. +### Rate Limiting + +**Cause:** Subscription quota exhausted. **Fix:** -- Fallback hinzufügen: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Verwenden Sie GLM/MiniMax als günstiges Backup### OAuth Token Expired +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -OmniRoute aktualisiert Token automatisch. Wenn die Probleme weiterhin bestehen: +### OAuth Token Expired -1. Dashboard → Anbieter → Erneut verbinden -2. Löschen Sie die Provider-Verbindung und fügen Sie sie erneut hinzu--- +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Überprüfen Sie, ob „BASE_URL“ auf Ihre laufende Instanz verweist (z. B. „http://localhost:20128“). -2. Überprüfen Sie, ob „CLOUD_URL“ auf Ihren Cloud-Endpunkt verweist (z. B. „https://omniroute.dev“). -3. Halten Sie die Werte von „NEXT*PUBLIC*\*“ an den serverseitigen Werten ausgerichtet### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Symptom:**„Unerwartetes Token „d“...“ auf dem Cloud-Endpunkt für Nicht-Streaming-Aufrufe. +### Cloud `stream=false` Returns 500 -**Ursache:**Upstream gibt SSE-Nutzdaten zurück, während der Client JSON erwartet. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Problemumgehung:**Verwenden Sie „stream=true“ für Cloud-Direktaufrufe. Die lokale Laufzeit umfasst SSE→JSON-Fallback.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Erstellen Sie einen neuen Schlüssel aus dem lokalen Dashboard („/api/keys“). -2. Führen Sie die Cloud-Synchronisierung aus: Cloud aktivieren → Jetzt synchronisieren -3. Alte/nicht synchronisierte Schlüssel können in der Cloud immer noch „401“ zurückgeben--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Überprüfen Sie die Laufzeitfelder: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. Für den tragbaren Modus: Verwenden Sie das Image-Ziel „runner-cli“ (gebündelte CLIs). -3. Für den Host-Mount-Modus: Legen Sie „CLI_EXTRA_PATHS“ fest und mounten Sie das Host-Bin-Verzeichnis als schreibgeschützt -4. Wenn „installed=true“ und „runnable=false“: Binärdatei wurde gefunden, aber die Integritätsprüfung ist fehlgeschlagen### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Überprüfen Sie die Nutzungsstatistiken im Dashboard → Nutzung -2. Primärmodell auf GLM/MiniMax umstellen -3. Nutzen Sie den kostenlosen Tarif (Gemini CLI, Qoder) für unkritische Aufgaben -4. Legen Sie Kostenbudgets pro API-Schlüssel fest: Dashboard → API-Schlüssel → Budget--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Setzen Sie „ENABLE_REQUEST_LOGS=true“ in Ihrer „.env“-Datei. Protokolle werden im Verzeichnis „logs/“ angezeigt.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,102 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Hauptstatus: „${DATA_DIR}/storage.sqlite“ (Anbieter, Kombinationen, Aliase, Schlüssel, Einstellungen) -- Verwendung: SQLite-Tabellen in „storage.sqlite“ („usage_history“, „call_logs“, „proxy_logs“) + optional „${DATA_DIR}/log.txt“ und „${DATA_DIR}/call_logs/“. -- Protokolle anfordern: `/logs/...` (wenn `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Wenn der Leistungsschalter eines Anbieters OFFEN ist, werden Anfragen blockiert, bis die Abklingzeit abgelaufen ist. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. **Fix:** -1. Gehen Sie zu**Dashboard → Einstellungen → Resilienz** -2. Überprüfen Sie die Leistungsschalterkarte des betroffenen Anbieters -3. Klicken Sie auf**Alle zurücksetzen**, um alle Unterbrecher zu löschen, oder warten Sie, bis die Abklingzeit abgelaufen ist -4. Stellen Sie vor dem Zurücksetzen sicher, dass der Anbieter tatsächlich verfügbar ist### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Wenn ein Anbieter wiederholt in den OPEN-Zustand wechselt: +### Provider keeps tripping the circuit breaker -1. Überprüfen Sie**Dashboard → Health → Provider Health**auf das Fehlermuster -2. Gehen Sie zu**Einstellungen → Ausfallsicherheit → Anbieterprofile**und erhöhen Sie den Fehlerschwellenwert -3. Überprüfen Sie, ob der Anbieter die API-Grenzwerte geändert hat oder eine erneute Authentifizierung erfordert -4. Überprüfen Sie die Latenz-Telemetrie – hohe Latenz kann zu zeitüberschreitungsbedingten Fehlern führen--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Stellen Sie sicher, dass Sie das richtige Präfix verwenden: „deepgram/nova-3“ oder „assemblyai/best“. -- Überprüfen Sie, ob der Anbieter unter**Dashboard → Anbieter**verbunden ist.### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Überprüfen Sie die unterstützten Audioformate: „mp3“, „wav“, „m4a“, „flac“, „ogg“, „webm“. -- Stellen Sie sicher, dass die Dateigröße innerhalb der Anbietergrenzen liegt (normalerweise < 25 MB). -- Überprüfen Sie die Gültigkeit des API-Schlüssels des Anbieters auf der Anbieterkarte--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Verwenden Sie**Dashboard → Übersetzer**, um Formatübersetzungsprobleme zu beheben: +Use **Dashboard → Translator** to debug format translation issues: -| Modus | Wann zu verwenden | -| ---------------- | --------------------------------------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Spielplatz** | Vergleichen Sie Eingabe-/Ausgabeformate nebeneinander – fügen Sie eine fehlgeschlagene Anfrage ein, um zu sehen, wie sie übersetzt wird | -| **Chat-Tester** | Senden Sie Live-Nachrichten und überprüfen Sie die vollständige Anfrage-/Antwort-Nutzlast einschließlich Header | -| **Prüfstand** | Führen Sie Stapeltests über Formatkombinationen hinweg durch, um herauszufinden, welche Übersetzungen fehlerhaft sind | -| **Live-Monitor** | Beobachten Sie den Anfragefluss in Echtzeit, um zeitweise auftretende Übersetzungsprobleme zu erkennen | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Thinking-Tags werden nicht angezeigt**– Überprüfen Sie, ob der Zielanbieter Thinking und die Einstellung des Thinking-Budgets unterstützt -**Tool-Aufrufe löschen**– Bei einigen Formatübersetzungen werden möglicherweise nicht unterstützte Felder entfernt. im Playground-Modus überprüfen -**Systemaufforderung fehlt**– Claude und Gemini gehen unterschiedlich mit Systemaufforderungen um; Überprüfen Sie die Übersetzungsausgabe -**SDK gibt Rohzeichenfolge statt Objekt zurück**– In Version 1.1.0 behoben: Antwortbereinigung entfernt jetzt nicht standardmäßige Felder (`x_groq`, `usage_breakdown` usw.), die zu OpenAI SDK Pydantic-Validierungsfehlern führen -**GLM/ERNIE lehnt „System“-Rolle ab**– In Version 1.1.0 behoben: Der Rollennormalisierer führt automatisch Systemmeldungen in Benutzermeldungen für inkompatible Modelle zusammen -**Rolle „Entwickler“ nicht erkannt**– In Version 1.1.0 behoben: Für Nicht-OpenAI-Anbieter automatisch in „System“ konvertiert -**`json_schema` funktioniert nicht mit Gemini**– In v1.1.0 behoben: `response_format` wird jetzt in Geminis `responseMimeType` + `responseSchema` konvertiert--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -– Die automatische Ratenbegrenzung gilt nur für API-Schlüsselanbieter (nicht OAuth/Abonnement). +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -- Überprüfen Sie, ob in**Einstellungen → Ausfallsicherheit → Anbieterprofile**die automatische Ratenbegrenzung aktiviert ist -- Überprüfen Sie, ob der Anbieter „429“-Statuscodes oder „Retry-After“-Header zurückgibt### Tuning exponential backoff +### Tuning exponential backoff -Anbieterprofile unterstützen diese Einstellungen: +Provider profiles support these settings: --**Basisverzögerung**– Anfängliche Wartezeit nach dem ersten Fehler (Standard: 1 s) -**Max. Verzögerung**– Maximale Wartezeitobergrenze (Standard: 30 s) -**Multiplikator**– Wie viel Verzögerung pro aufeinanderfolgendem Fehler erhöht werden soll (Standard: 2x)### Anti-thundering herd +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) -Wenn viele gleichzeitige Anfragen einen Anbieter mit begrenzter Rate treffen, verwendet OmniRoute Mutex + automatische Ratenbegrenzung, um Anfragen zu serialisieren und kaskadierende Fehler zu verhindern. Dies geschieht automatisch für API-Schlüsselanbieter.--- +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Einige OmniRoute-Benutzer platzieren das Gateway vor RAG- oder Agent-Stacks. In diesen Setups ist es üblich, ein seltsames Muster zu erkennen: OmniRoute sieht fehlerfrei aus (Anbieter aktiv, Routing-Profile in Ordnung, keine Ratenbegrenzungswarnungen), aber die endgültige Antwort ist immer noch falsch. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -In der Praxis gehen diese Vorfälle meist von der nachgelagerten RAG-Pipeline aus, nicht vom Gateway selbst. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Wenn Sie ein gemeinsames Vokabular zur Beschreibung dieser Fehler wünschen, können Sie die WFGY ProblemMap verwenden, eine externe MIT-Lizenztextressource, die sechzehn wiederkehrende RAG-/LLM-Fehlermuster definiert. Auf hohem Niveau umfasst es: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- Abrufdrift und gebrochene Kontextgrenzen -- leere oder veraltete Indizes und Vektorspeicher -- Einbettung versus semantische Nichtübereinstimmung -- Probleme mit der Eingabeaufforderung und dem Kontextfenster -- Zusammenbruch der Logik und übertriebene Antworten -- Fehler bei der Koordinierung langer Ketten und Agenten -- Multiagentengedächtnis und Rollendrift -- Probleme bei der Bereitstellung und Bootstrap-Reihenfolge +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -Die Idee ist einfach: +The idea is simple: -1. Wenn Sie eine schlechte Antwort untersuchen, erfassen Sie Folgendes: - - Benutzeraufgabe und -anfrage - - Routen- oder Anbieterkombination in OmniRoute - - jeglicher RAG-Kontext, der nachgelagert verwendet wird (abgerufene Dokumente, Tool-Aufrufe usw.) -2. Ordnen Sie den Vorfall einer oder zwei WFGY ProblemMap-Nummern („Nr. 1“ … „Nr. 16“) zu. -3. Speichern Sie die Nummer in Ihrem eigenen Dashboard, Runbook oder Incident-Tracker neben den OmniRoute-Protokollen. -4. Verwenden Sie die entsprechende WFGY-Seite, um zu entscheiden, ob Sie Ihren RAG-Stack, Retriever oder Ihre Routing-Strategie ändern müssen. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -Volltext und konkrete Rezepte gibt es hier (MIT-Lizenz, nur Text): +Full text and concrete recipes live here (MIT license, text only): [WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -Sie können diesen Abschnitt ignorieren, wenn Sie keine RAG- oder Agent-Pipelines hinter OmniRoute ausführen.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**GitHub-Probleme**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Architektur**: Interne Details finden Sie unter [`docs/ARCHITECTURE.md`](ARCHITECTURE.md). -**API-Referenz**: Siehe [`docs/API_REFERENCE.md`](API_REFERENCE.md) für alle Endpunkte -**Gesundheits-Dashboard**: Überprüfen Sie**Dashboard → Gesundheit**auf den Echtzeit-Systemstatus -**Übersetzer**: Verwenden Sie**Dashboard → Übersetzer**, um Formatprobleme zu beheben +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/de/llm.txt b/docs/i18n/de/llm.txt new file mode 100644 index 0000000000..82c5317d30 --- /dev/null +++ b/docs/i18n/de/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Deutsch) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Übersicht + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Sicherheit +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/es/README.md b/docs/i18n/es/README.md index ced39b01f1..47720abff4 100644 --- a/docs/i18n/es/README.md +++ b/docs/i18n/es/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Su proxy API universal: un punto final, más de 60 proveedores, cero tiempo de inactividad. Ahora con**Servidor MCP (25 herramientas)**,**Protocolo A2A**,**Sistemas de memoria/habilidades**y**Aplicación de escritorio Electron**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Finalización de chat • Incrustaciones • Generación de imágenes • Vídeo • Música • Audio • Reclasificación •**Búsqueda web**• Servidor MCP • Protocolo A2A • 100% TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Su proxy API universal: un punto final, más de 60 proveedores, cero tiempo de [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Sitio web](https://omniroute.online) • [🚀 Inicio rápido](#-inicio rápido) • [💡 Funciones](#-key-features) • [📖 Documentos](#-documentación) • [💰 Precios](#-precios-de-un-vistazo) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Disponible en:**🇺🇸 [Inglés](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [English](docs/i18n/es/README.md) | 🇫🇷 [Francés](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magiar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Países Bajos](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Esloveno](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,553 +60,629 @@ _Su proxy API universal: un punto final, más de 60 proveedores, cero tiempo de ## 📸 Dashboard Preview - -Haga clic para ver capturas de pantalla del panel +
+Click to see dashboard screenshots -| Página | Captura de pantalla | -| -------------------- | ------------------------------------------------------ | ---------- | -| **Proveedores** | ![Proveedores](docs/screenshots/01-providers.png) | -| **Combinaciones** | ![Combos](docs/screenshots/02-combos.png) | -| **Análisis** | ![Análisis](docs/screenshots/03-analytics.png) | -| **Salud** | ![Salud](docs/screenshots/04-health.png) | -| **Traductor** | ![Traductor](docs/screenshots/05-translator.png) | -| **Configuración** | ![Configuración](docs/screenshots/06-settings.png) | -| **Herramientas CLI** | ![Herramientas CLI](docs/screenshots/07-cli-tools.png) | -| **Registros de uso** | ![Uso](docs/screenshots/08-usage.png) | -| **Puntos finales** | ![Puntos finales](docs/screenshots/09-endpoint.png) |
| +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | + +
--- ### 🤖 Free AI Provider for your favorite coding agents -_Conecte cualquier herramienta IDE o CLI con tecnología de IA a través de OmniRoute: puerta de enlace API gratuita para codificación ilimitada._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ - + - - - - - - - - - - - +
+ - OpenClaw>
+ OpenClaw
OpenClaw

- ⭐205K + ⭐ 205K
+ - NanoBot>
- Nanobot + NanoBot
+ NanoBot

- ⭐ 20,9K + ⭐ 20.9K
+ - PicoClaw>
- PicoGarra + PicoClaw
+ PicoClaw

- ⭐ 14,6K + ⭐ 14.6K
+ - ZeroClaw>
- Garra Cero + ZeroClaw
+ ZeroClaw

- ⭐ 9,9K + ⭐ 9.9K
+ - IronClaw>
- Garra de Hierro + IronClaw
+ IronClaw

- ⭐ 2,1K + ⭐ 2.1K
+ - OpenCode>
- Código abierto + OpenCode
+ OpenCode

⭐ 106K
+ - Codex CLI>
- CLI del Códice + Codex CLI
+ Codex CLI

- ⭐ 60,8K + ⭐ 60.8K
+ - Código Claude>
- Código Claude + Claude Code
+ Claude Code

- ⭐ 67,3K + ⭐ 67.3K
+ - Gemini CLI>
- CLI de Géminis + Gemini CLI
+ Gemini CLI

- ⭐ 94,7K + ⭐ 94.7K
+ - Código Kilo>
- Código Kilo + Kilo Code
+ Kilo Code

- ⭐ 15,5K + ⭐ 15.5K
-📡 Todos los agentes se conectan a través de http://localhost:20128/v1 o http://cloud.omniroute.online/v1: una configuración, modelos y cuotas ilimitados--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Deja de gastar dinero y alcanzar límites:** +**Stop wasting money and hitting limits:** -- La cuota de suscripción vence cada mes sin usarse -- Los límites de velocidad le impiden codificar a mitad de camino -- API costosas ($20-50/mes por proveedor) -- Cambio manual entre proveedores +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute resuelve esto:** +**OmniRoute solves this:** -- ✅**Maximizar suscripciones**- Realice un seguimiento de la cuota, use cada bit antes de restablecer -- ✅**Retroceso automático**- Suscripción → Clave API → Barato → Gratis, sin tiempo de inactividad -- ✅**Multicuenta**- Round-robin entre cuentas por proveedor -- ✅**Universal**- Funciona con Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw y cualquier herramienta CLI--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**¡Únase a nuestra comunidad!**[Grupo de WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t): obtenga ayuda, comparta consejos y manténgase actualizado. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Sitio web**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Problemas**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Grupo comunitario](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Contribuyendo**: consulte [CONTRIBUTING.md](CONTRIBUTING.md), abra un PR o elija un "buen primer número". -**Proyecto original**: [9router de decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Al abrir un problema, ejecute el comando system-info y adjunte el archivo generado:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Esto genera un `system-info.txt` con su versión de Node.js, versión de OmniRoute, detalles del sistema operativo, herramientas CLI instaladas (qoder, gemini, claude, codex, antigravity, droid, etc.), estado de Docker/PM2 y paquetes del sistema: todo lo que necesitamos para reproducir su problema rápidamente. Adjunte el archivo directamente a su problema de GitHub.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Todos los desarrolladores que utilizan herramientas de IA se enfrentan a estos problemas a diario.**OmniRoute se creó para resolverlos todos: desde sobrecostos hasta bloqueos regionales, desde flujos rotos de OAuth hasta operaciones de protocolo y observabilidad empresarial. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. - -💸 1. "Pago una suscripción costosa pero aún así me interrumpen los límites" +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Los desarrolladores pagan entre 20 y 200 dólares al mes por Claude Pro, Codex Pro o GitHub Copilot. Incluso pagando, la cuota tiene un límite: 5 horas de uso, límites semanales o límites de tarifa por minuto. A mitad de la sesión de codificación, el proveedor deja de responder y el desarrollador pierde flujo y productividad. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Cómo lo resuelve OmniRoute:** +**How OmniRoute solves it:** --**Reserva inteligente de 4 niveles**: si se agota la cuota de suscripción, se redirige automáticamente a la clave API → Barato → Gratis sin intervención manual --**Seguimiento de límites del proveedor**: las instantáneas de cuota almacenadas en caché se actualizan según una programación del lado del servidor (predeterminado `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) con actualización manual disponible en la interfaz de usuario --**Soporte multicuenta**: varias cuentas por proveedor con rotación automática: cuando una se agota, cambia a la siguiente --**Combinaciones personalizadas**: cadenas de respaldo personalizables con 9 estrategias de equilibrio (prioridad, ponderada, llenado primero, round-robin, P2C, aleatoria, menos utilizada, de costo optimizado, estrictamente aleatoria) --**Cuotas comerciales de Codex**: monitoreo de cuotas del espacio de trabajo empresarial/de equipo directamente en el panel
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard - -🔌 2. "Necesito usar varios proveedores pero cada uno tiene una API diferente" +
-OpenAI usa un formato, Claude (Anthropic) usa otro, Gemini otro más. Si un desarrollador quiere probar modelos de diferentes proveedores o recurrir a ellos, debe reconfigurar los SDK, cambiar los puntos finales y lidiar con formatos incompatibles. Los proveedores personalizados (FriendLI, NIM) tienen puntos finales de modelo no estándar. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Cómo lo resuelve OmniRoute:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Punto final unificado**: un único `http://localhost:20128/v1` sirve como proxy para los más de 60 proveedores --**Traducción de formato**: automática y transparente: OpenAI ↔ Claude ↔ Gemini ↔ API de respuestas --**Desinfección de respuesta**: elimina los campos no estándar (`x_groq`, `usage_breakdown`, `service_tier`) que interrumpen OpenAI SDK v1.83+ --**Normalización de roles**: convierte `desarrollador` → `sistema` para proveedores que no son OpenAI; `sistema` → `usuario` para GLM/ERNIE --**Think Tag Extraction**: extrae bloques `` de modelos como DeepSeek R1 en `reasoning_content` estandarizado. --**Salida estructurada para Gemini**— conversión automática `json_schema` → `responseMimeType`/`responseSchema` --**`stream` por defecto es `false`**: se alinea con las especificaciones de OpenAI, evitando SSE inesperado en los SDK de Python/Rust/Go
+**How OmniRoute solves it:** - -🌐 3. "Mi proveedor de IA bloquea mi región/país" +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Proveedores como OpenAI/Codex bloquean el acceso desde ciertas regiones geográficas. Los usuarios reciben errores como `unsupported_country_region_territory` durante las conexiones OAuth y API. Esto resulta especialmente frustrante para los desarrolladores de los países en desarrollo. +
-**Cómo lo resuelve OmniRoute:** +
+🌐 3. "My AI provider blocks my region/country" --**Configuración de proxy de 3 niveles**: Proxy configurable en 3 niveles: global (todo el tráfico), por proveedor (un solo proveedor) y por conexión/clave. --**Insignias de proxy codificadas por colores**— Indicadores visuales: 🟢 proxy global, 🟡 proxy de proveedor, 🔵 proxy de conexión, que siempre muestra la IP --**Intercambio de tokens de OAuth a través de proxy**: el flujo de OAuth también pasa a través del proxy, lo que resuelve `unsupported_country_region_territory` --**Pruebas de conexión a través de proxy**: las pruebas de conexión utilizan el proxy configurado (no más derivación directa) --**Soporte SOCKS5**: soporte completo de proxy SOCKS5 para enrutamiento saliente --**Suplantación de huellas dactilares TLS**: huella digital TLS similar a la de un navegador a través de `wreq-js` para evitar la detección de bots --**🔏 Coincidencia de huellas dactilares CLI**: reordena los encabezados y los campos del cuerpo para que coincidan con las firmas binarias CLI nativas, lo que reduce drásticamente el riesgo de marcación de cuentas. La IP del proxy se conserva: obtienes enmascaramiento de IP oculto**y**simultáneamente
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. - -🆓 4. "Quiero usar IA para codificar pero no tengo dinero" +**How OmniRoute solves it:** -No todo el mundo puede pagar entre 20 y 200 dólares al mes por suscripciones a IA. Los estudiantes, desarrolladores de países emergentes, aficionados y autónomos necesitan acceso a modelos de calidad sin coste alguno. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Cómo lo resuelve OmniRoute:** +
--**Proveedores de nivel gratuito integrados**: soporte nativo para proveedores 100 % gratuitos: Qoder (5 modelos ilimitados a través de OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 modelos ilimitados: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID gratis), Gemini CLI (180.000 tokens/mes gratis) --**Ollama Cloud**: modelos de Ollama alojados en la nube en `api.ollama.com` con nivel gratuito de "Uso ligero"; use el prefijo `ollamacloud/` --**Combos solo gratuitos**— Cadena `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/mes sin tiempo de inactividad --**Acceso gratuito a NVIDIA NIM**: desarrollo de ~40 RPM, acceso gratuito para siempre a más de 70 modelos en build.nvidia.com (transición de créditos a límites de velocidad pura) --**Estrategia de optimización de costos**: estrategia de enrutamiento que elige automáticamente el proveedor más barato disponible
+
+🆓 4. "I want to use AI for coding but I have no money" - -🔒 5. "Necesito proteger mi puerta de enlace de IA del acceso no autorizado" +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -Al exponer una puerta de enlace de IA a la red (LAN, VPS, Docker), cualquiera con la dirección puede consumir los tokens/cuota del desarrollador. Sin protección, las API son vulnerables al mal uso, la inyección rápida y el abuso. +**How OmniRoute solves it:** -**Cómo lo resuelve OmniRoute:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**Administración de claves API**: generación, rotación y alcance por proveedor con una página dedicada `/dashboard/api-manager` --**Permisos a nivel de modelo**: restrinja las claves API a modelos específicos (`openai/*`, patrones comodín), con la opción Permitir todo/Restringir --**API Endpoint Protection**: requiere una clave para `/v1/models` y bloquea proveedores específicos del listado --**Auth Guard + Protección CSRF**: todas las rutas del panel protegidas con middleware `withAuth` + tokens CSRF --**Limitador de velocidad**: limitación de velocidad por IP con ventanas configurables --**Filtrado de IP**: lista permitida/lista bloqueada para control de acceso --**Prompt injection guard**: desinfección contra patrones de avisos maliciosos --**Cifrado AES-256-GCM**: credenciales cifradas en reposo
+
- -🛑 6. "Mi proveedor dejó de funcionar y perdí mi flujo de codificación" +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -Los proveedores de IA pueden volverse inestables, devolver errores 5xx o alcanzar límites de velocidad temporales. Si un desarrollador depende de un solo proveedor, se le interrumpe. Sin disyuntores, los reintentos repetidos pueden bloquear la aplicación. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Cómo lo resuelve OmniRoute:** +**How OmniRoute solves it:** --**Disyuntor por modelo**: apertura/cierre automático con umbrales configurables y enfriamiento (cerrado/abierto/medio abierto), con alcance por modelo para evitar bloqueos en cascada --**Retroceso exponencial**: retrasos progresivos en los reintentos --**Anti-Thundering Herd**— Mutex + protección de semáforo contra tormentas de reintentos simultáneos --**Cadenas alternativas combinadas**: si el proveedor principal falla, automáticamente pasa por la cadena sin intervención. --**Disyuntor combinado**: desactiva automáticamente los proveedores defectuosos dentro de una cadena combinada --**Panel de estado**: monitoreo del tiempo de actividad, estados de disyuntores, bloqueos, estadísticas de caché, latencia p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest - -🔧 7. "Configurar cada herramienta de IA es tedioso y repetitivo" + -Los desarrolladores utilizan Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Cada herramienta necesita una configuración diferente (punto final API, clave, modelo). Reconfigurar al cambiar de proveedor o modelo es una pérdida de tiempo. +
+🛑 6. "My provider went down and I lost my coding flow" -**Cómo lo resuelve OmniRoute:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**Panel de herramientas CLI**: página dedicada con configuración con un solo clic para Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline --**Generador de configuración de GitHub Copilot**: genera `chatLanguageModels.json` para código VS con selección masiva de modelos --**Asistente de incorporación**: configuración guiada de 4 pasos para usuarios nuevos --**Un punto final, todos los modelos**: configure `http://localhost:20128/v1` una vez, acceda a más de 60 proveedores
+**How OmniRoute solves it:** - -🔑 8. "Administrar tokens OAuth de múltiples proveedores es un infierno" +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot: todos usan OAuth 2.0 con tokens que caducan. Los desarrolladores necesitan volver a autenticarse constantemente, lidiar con "falta client_secret", "redirect_uri_mismatch" y fallas en servidores remotos. OAuth en LAN/VPS es particularmente problemático. + -**Cómo lo resuelve OmniRoute:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Actualización automática de tokens**: los tokens de OAuth se actualizan en segundo plano antes de que caduquen --**OAuth 2.0 (PKCE) integrado**: flujo automático para Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder --**OAuth multicuenta**: varias cuentas por proveedor mediante extracción de token JWT/ID --**OAuth LAN/Remote Fix**— Detección de IP privada para `redirect_uri` + modo URL manual para servidores remotos --**OAuth detrás de Nginx**: utiliza `window.location.origin` para compatibilidad con proxy inverso --**Guía remota de OAuth**: guía paso a paso para las credenciales de Google Cloud en VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. - -📊 9. "No sé cuánto estoy gastando ni dónde" +**How OmniRoute solves it:** -Los desarrolladores utilizan múltiples proveedores pagos pero no tienen una visión unificada del gasto. Cada proveedor tiene su propio panel de facturación, pero no hay una vista consolidada. Los costos inesperados pueden acumularse. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Cómo lo resuelve OmniRoute:** + --**Panel de análisis de costos**: seguimiento de costos por token y gestión de presupuesto por proveedor --**Límites de presupuesto por nivel**: límite de gasto por nivel que activa el respaldo automático --**Configuración de precios por modelo**: precios configurables por modelo --**Estadísticas de uso por clave API**: recuento de solicitudes y marca de tiempo utilizada por última vez por clave --**Panel de análisis**: tarjetas de estadísticas, tabla de uso de modelos, tabla de proveedores con tasas de éxito y latencia. +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" - -🐛 10. "No puedo diagnosticar errores ni problemas en las llamadas de IA" +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Cuando falla una llamada, el desarrollador no sabe si se trata de un límite de velocidad, un token caducado, un formato incorrecto o un error del proveedor. Registros fragmentados en diferentes terminales. Sin observabilidad, la depuración es de prueba y error. +**How OmniRoute solves it:** -**Cómo lo resuelve OmniRoute:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Panel de registros unificados**: 4 pestañas: registros de solicitudes, registros de proxy, registros de auditoría y consola --**Visor de registros de consola**: visor estilo terminal en tiempo real con niveles codificados por colores, desplazamiento automático, búsqueda y filtro --**Registros de proxy SQLite**: registros persistentes que sobreviven a los reinicios del servidor --**Translator Playground**: 4 modos de depuración: Playground (traducción de formato), Chat Tester (ida y vuelta), Test Bench (por lotes), Live Monitor (en tiempo real) --**Solicitud de telemetría**: latencia p50/p95/p99 + seguimiento de X-Request-Id --**Registro basado en archivos con rotación**: los registros de aplicaciones rotan por tamaño, días de retención y recuento de archivos; Los artefactos del registro de llamadas rotan según los días de retención y el recuento de archivos. --**Informe de información del sistema**: `npm run system-info` genera `system-info.txt` con su entorno completo (versión de nodo, versión de OmniRoute, sistema operativo, herramientas CLI, estado de Docker/PM2). Adjúntelo cuando informe problemas para una clasificación instantánea.
+ - -🏗️ 11. "Implementar y mantener la puerta de enlace es complejo" +
+📊 9. "I don't know how much I'm spending or where" -Instalar, configurar y mantener un proxy de IA en diferentes entornos (local, VPS, Docker, nube) requiere mucha mano de obra. Problemas como rutas codificadas, "EACCES" en directorios, conflictos de puertos y compilaciones multiplataforma añaden fricción. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Cómo lo resuelve OmniRoute:** +**How OmniRoute solves it:** --**npm global install**— `npm install -g omniroute && omniroute` — hecho --**Docker multiplataforma**: AMD64 + ARM64 nativo (Apple Silicon, AWS Graviton, Raspberry Pi) --**Docker Compose Profiles**— `base` (sin herramientas CLI) y `cli` (con Claude Code, Codex, OpenClaw) --**Aplicación de escritorio Electron**: aplicación nativa para Windows/macOS/Linux con bandeja del sistema, inicio automático y modo sin conexión --**Modo de puerto dividido**: API y panel en puertos separados para escenarios avanzados (proxy inverso, redes de contenedores) --**Cloud Sync**: sincronización de configuración entre dispositivos a través de Cloudflare Workers --**Copias de seguridad de base de datos**: copia de seguridad, restauración, exportación e importación automáticas de todas las configuraciones, con `DISABLE_SQLITE_AUTO_BACKUP` para copias de seguridad administradas externamente
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency - -🌍 12. "La interfaz es solo en inglés y mi equipo no habla inglés" + -Los equipos en países que no hablan inglés, especialmente en América Latina, Asia y Europa, tienen dificultades con las interfaces solo en inglés. Las barreras del idioma reducen la adopción y aumentan los errores de configuración. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Cómo lo resuelve OmniRoute:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Panel i18n — 30 idiomas**— Las más de 500 teclas traducidas, incluidas árabe, búlgaro, danés, alemán, español, finlandés, francés, hebreo, hindi, húngaro, indonesio, italiano, japonés, coreano, malayo, holandés, noruego, polaco, portugués (PT/BR), rumano, ruso, eslovaco, sueco, tailandés, ucraniano, vietnamita, chino, filipino, inglés. --**Soporte RTL**: soporte de derecha a izquierda para árabe y hebreo --**README multilingüe**: 30 traducciones de documentación completa --**Selector de idioma**: ícono de globo en el encabezado para cambiar en tiempo real
+**How OmniRoute solves it:** - -🔄 13. "Necesito más que chat: necesito incrustaciones, imágenes y audio" +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -La IA no es solo completar un chat. Los desarrolladores necesitan generar imágenes, transcribir audio, crear incrustaciones para RAG, reclasificar documentos y moderar contenido. Cada API tiene un punto final y un formato diferentes. + -**Cómo lo resuelve OmniRoute:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Incrustaciones**— `/v1/embeddings` con 6 proveedores y más de 9 modelos --**Generación de imágenes**— `/v1/images/generaciones` con 10 proveedores y más de 20 modelos (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Texto a vídeo**— `/v1/videos/generaciones` — ComfyUI (AnimateDiff, SVD) y SD WebUI --**Texto a música**— `/v1/music/generaciones` — ComfyUI (Audio estable abierto, MusicGen) --**Transcripción de audio**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Text-to-Speech**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + proveedores existentes --**Moderaciones**— `/v1/moderaciones` — Comprobaciones de seguridad del contenido --**Reclasificación**— `/v1/rerank` — Reclasificación de relevancia del documento --**API de respuestas**: compatibilidad total con `/v1/responses` para Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. - -🧪 14. "No tengo forma de probar y comparar la calidad entre modelos" +**How OmniRoute solves it:** -Los desarrolladores quieren saber qué modelo es mejor para su caso de uso (código, traducción, razonamiento), pero comparar manualmente es lento. No existen herramientas de evaluación integradas. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**Cómo lo resuelve OmniRoute:** + --**Evaluaciones LLM**: pruebas de conjunto dorado con 10 casos precargados que cubren saludos, matemáticas, geografía, generación de código, cumplimiento de JSON, traducción, rebajas y rechazo de seguridad. --**4 estrategias de coincidencia**: `exact`, `contains`, `regex`, `custom` (función JS) --**Translator Playground Test Bench**: pruebas por lotes con múltiples entradas y resultados esperados, comparación entre proveedores --**Chat Tester**: recorrido completo de ida y vuelta con representación de respuesta visual --**Live Monitor**: flujo en tiempo real de todas las solicitudes que fluyen a través del proxy +
+🌍 12. "The interface is English-only and my team doesn't speak English" - -📈 15. "Necesito escalar sin perder rendimiento" +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -A medida que crece el volumen de solicitudes, sin almacenar en caché las mismas preguntas generan costos duplicados. Sin idempotencia, las solicitudes duplicadas desperdician el procesamiento. Se deben respetar los límites de tarifas por proveedor. +**How OmniRoute solves it:** -**Cómo lo resuelve OmniRoute:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Caché semántica**: la caché de dos niveles (firma + semántica) reduce el costo y la latencia --**Solicitud de idempotencia**: ventana de deduplicación de 5 segundos para solicitudes idénticas --**Detección de límite de velocidad**: RPM por proveedor, intervalo mínimo y seguimiento simultáneo máximo --**Límites de velocidad editables**: valores predeterminados configurables en Configuración → Resiliencia con persistencia --**Caché de validación de clave API**: caché de 3 niveles para rendimiento de producción --**Panel de estado con telemetría**: latencia p50/p95/p99, estadísticas de caché, tiempo de actividad
+ - -🤖 16. "Quiero controlar el comportamiento del modelo globalmente" +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Desarrolladores que quieran todas las respuestas en un idioma específico, con un tono específico o quieran limitar los tokens de razonamiento. Configurar esto en cada herramienta/solicitud no es práctico. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Cómo lo resuelve OmniRoute:** +**How OmniRoute solves it:** --**Inyección de aviso del sistema**: aviso global aplicado a todas las solicitudes --**Thinking Budget Validation**: control de asignación de tokens de razonamiento por solicitud (transferencia, automática, personalizada, adaptativa) --**9 estrategias de enrutamiento**: estrategias globales que determinan cómo se distribuyen las solicitudes --**Enrutador comodín**: los patrones `proveedor/*` se enrutan dinámicamente a cualquier proveedor --**Activar/desactivar combinación de alternar**: alterna combinaciones directamente desde el panel --**Alternar proveedor**: activa/desactiva todas las conexiones de un proveedor con un solo clic --**Proveedores bloqueados**: excluye proveedores específicos de la lista `/v1/models`
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex - -🧰 17. "Necesito herramientas MCP como capacidades de producto de primera clase" + -Muchas puertas de enlace de IA exponen MCP solo como un detalle de implementación oculto. Los equipos necesitan una capa operativa visible y manejable. +
+🧪 14. "I have no way to test and compare quality across models" -**Cómo lo resuelve OmniRoute:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP aparece en la pestaña de navegación del panel y protocolo de punto final -- Página de gestión de MCP dedicada con procesos, herramientas, alcances y auditoría -- Inicio rápido integrado para `omniroute --mcp` e incorporación de clientes
+**How OmniRoute solves it:** - -🧠 18. "Necesito orquestación A2A con rutas de tareas de sincronización y transmisión" +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Los flujos de trabajo de los agentes necesitan respuestas directas y una ejecución continua y continua con control del ciclo de vida. + -**Cómo lo resuelve OmniRoute:** +
+📈 15. "I need to scale without losing performance" -- Punto final A2A JSON-RPC (`POST /a2a`) con `mensaje/envío` y `mensaje/transmisión` -- Transmisión SSE con propagación del estado terminal -- API de ciclo de vida de tareas para `tareas/obtener` y `tareas/cancelar`
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. - -🛰️ 19. "Necesito un estado real del proceso MCP, no un estado adivinado" +**How OmniRoute solves it:** -Los equipos operativos necesitan saber si MCP está realmente activo, no solo si se puede acceder a una API. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**Cómo lo resuelve OmniRoute:** + -- Archivo de latidos en tiempo de ejecución con PID, marcas de tiempo, transporte, recuento de herramientas y modo de alcance -- API de estado de MCP que combina latidos + actividad reciente -- Tarjetas de estado de la interfaz de usuario para el proceso/tiempo de actividad/actualización de latidos +
+🤖 16. "I want to control model behavior globally" - -📋 20. "Necesito ejecución de herramienta MCP auditable" +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Cuando las herramientas modifican la configuración o desencadenan acciones de operaciones, los equipos necesitan trazabilidad forense. +**How OmniRoute solves it:** -**Cómo lo resuelve OmniRoute:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- Registro de auditoría respaldado por SQLite para llamadas a herramientas MCP -- Filtros por herramienta, éxito/fracaso, clave API y paginación -- Tabla de auditoría del panel + puntos finales de estadísticas para automatización
+ - -🔐 21. "Necesito permisos MCP con alcance por integración" +
+🧰 17. "I need MCP tools as first-class product capabilities" -Los diferentes clientes deberían tener acceso con privilegios mínimos a las categorías de herramientas. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Cómo lo resuelve OmniRoute:** +**How OmniRoute solves it:** -- 10 alcances MCP granulares para acceso controlado a herramientas -- Aplicación del alcance y visibilidad en la interfaz de usuario de gestión de MCP -- Postura predeterminada segura para herramientas operativas
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding - -⚙️ 22. "Necesito controles operativos sin redistribuir" + -Los equipos necesitan cambios rápidos en el tiempo de ejecución durante incidentes o eventos de costos. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**Cómo lo resuelve OmniRoute:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Cambie la activación combinada directamente desde el panel de MCP -- Aplicar perfiles de resiliencia de paquetes de políticas predefinidos -- Restablecer el estado del disyuntor desde el mismo panel de operaciones.
+**How OmniRoute solves it:** - -🔄 23. "Necesito visibilidad y cancelación del ciclo de vida de la tarea A2A en vivo" +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Sin visibilidad del ciclo de vida, los incidentes de tareas se vuelven difíciles de clasificar. + -**Cómo lo resuelve OmniRoute:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Listado de tareas/filtrado por estado/habilidad con paginación -- Profundización en metadatos, eventos y artefactos de tareas -- Punto final de cancelación de tarea y acción de UI con confirmación
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. - -🌊 24. "Necesito métricas de transmisión activas para la carga A2A" +**How OmniRoute solves it:** -Los flujos de trabajo de streaming requieren información operativa sobre la simultaneidad y las conexiones en vivo. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**Cómo lo resuelve OmniRoute:** + -- Contadores de flujo activos integrados en el estado A2A -- Marca de tiempo de la última tarea y recuentos por estado -- Tarjetas de tablero A2A para monitoreo de operaciones en tiempo real +
+📋 20. "I need auditable MCP tool execution" - -🪪 25. "Necesito un descubrimiento de agentes estándar para los clientes" +When tools mutate config or trigger ops actions, teams need forensic traceability. -Los clientes y orquestadores externos necesitan metadatos legibles por máquina para la incorporación. +**How OmniRoute solves it:** -**Cómo lo resuelve OmniRoute:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- Tarjeta de agente expuesta en `/.well-known/agent.json` -- Capacidades y habilidades mostradas en la interfaz de usuario de gestión. -- La API de estado A2A incluye metadatos de descubrimiento para la automatización
+ - -🧭 26. "Necesito capacidad de descubrimiento del protocolo en la UX del producto" +
+🔐 21. "I need scoped MCP permissions per integration" -Si los usuarios no pueden descubrir las superficies de protocolo, la calidad de la adopción y el soporte disminuye. +Different clients should have least-privilege access to tool categories. -**Cómo lo resuelve OmniRoute:** +**How OmniRoute solves it:** -- Página consolidada de**Puntos finales**con pestañas para Proxy, MCP, A2A y API Endpoints -- El estado del servicio en línea alterna (en línea/fuera de línea) para MCP y A2A -- Enlaces desde la descripción general a pestañas de administración dedicadas
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling - -🧪 27. "Necesito validación de protocolo de un extremo a otro con clientes reales" + -Las pruebas simuladas no son suficientes para validar la compatibilidad del protocolo antes del lanzamiento. +
+⚙️ 22. "I need operational controls without redeploying" -**Cómo lo resuelve OmniRoute:** +Teams need quick runtime changes during incidents or cost events. -- Suite E2E que inicia la aplicación y utiliza transporte de cliente MCP SDK real -- Pruebas de cliente A2A para descubrimiento, envío, transmisión, obtención y cancelación de flujos -- Verificar las afirmaciones con las API de auditoría MCP y tareas A2A.
+**How OmniRoute solves it:** - -📡 28. "Necesito observabilidad unificada en todas las interfaces" +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -Dividir la observabilidad por protocolo crea puntos ciegos y MTTR más largos. + -**Cómo lo resuelve OmniRoute:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Paneles/registros/análisis unificados en un solo producto -- Salud + auditoría + solicitud de telemetría en capas OpenAI, MCP y A2A -- API operativas para estado y automatización.
+Without lifecycle visibility, task incidents become hard to triage. - -💼 29. "Necesito un tiempo de ejecución para proxy + herramientas + orquestación de agentes" +**How OmniRoute solves it:** -La ejecución de muchos servicios separados aumenta los costos operativos y los modos de falla. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**Cómo lo resuelve OmniRoute:** + -- Proxy compatible con OpenAI, servidor MCP y servidor A2A en una sola pila -- Autenticación compartida, resiliencia, almacenamiento de datos y observabilidad. -- Modelo de política consistente en todas las superficies de interacción. +
+🌊 24. "I need active stream metrics for A2A load" - -🚀 30. "Necesito enviar flujos de trabajo agentes sin expansión de códigos adhesivos" +Streaming workflows require operational insight into concurrency and live connections. -Los equipos pierden velocidad al unir múltiples scripts y servicios ad hoc. +**How OmniRoute solves it:** -**Cómo lo resuelve OmniRoute:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Estrategia de endpoint unificada para clientes y agentes -- UI de gestión de protocolos integradas y rutas de validación de humo -- Fundamentos listos para producción (seguridad, registro, resiliencia, respaldo)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Libro de estrategias A: maximizar la suscripción paga + copia de seguridad económica**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -607,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Libro de estrategias B: pila de codificación de costo cero**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Libro de estrategias C: cadena alternativa siempre disponible las 24 horas del día, los 7 días de la semana**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -630,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Libro de jugadas D: Operaciones del agente con MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Configure la codificación AI en minutos a**$0/mes**. Conecte estas cuentas gratuitas y utilice el combo**Free Stack**integrado. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Paso | Acción | Proveedores desbloqueados | +| Step | Action | Providers Unlocked | | ---- | -------------------------------------------------- | ------------------------------------------------------------------ | -| 1 | Conectar**Kiro**(ID de AWS Builder OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**ilimitado**| -| 2 | Conectar**Qoder**(Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... —**ilimitado**| -| 3 | Conectar**Qwen**(Código de dispositivo) | qwen3-coder-plus, qwen3-coder-flash... —**ilimitado**| -| 4 | Conectar**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180K/mes gratis**| -| 5 | `/dashboard/combos` →**Plantilla de pila gratuita ($0)**| Round-robin todos los proveedores gratuitos automáticamente | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Apunte cualquier IDE/CLI a:**`http://localhost:20128/v1` · Clave API: `any-string` · Listo. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Cobertura adicional opcional (también gratuita):**Clave API Groq (30 RPM gratis), NVIDIA NIM (40 RPM gratis, más de 70 modelos), Cerebras (1 millón de tok/día), clave API LongCat (¡50 millones de tokens/día!), Cloudflare Workers AI (10 000 neuronas/día, más de 50 modelos).## Inicio Rápido +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Inicio Rápido ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **usuarios de pnpm:**Ejecute `pnpm aprobar-builds -g` después de la instalación para habilitar los scripts de compilación nativos requeridos por `better-sqlite3` y `@swc/core`: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > -> ```golpecito -> pnpm instalar -g omniruta -> pnpm aprobar-builds -g # Seleccionar todos los paquetes → aprobar -> omniruta +> ```bash +> pnpm install -g omniroute +> pnpm approve-builds -g # Select all packages → approve +> omniroute > ``` -El panel se abre en `http://localhost:20128` y la URL base de API es `http://localhost:20128/v1`. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Comando | Descripción | -| ------------------------ | --------------------------------------------------------------- | -| `omniruta` | Iniciar servidor (`PORT=20128`, API y panel en el mismo puerto) | -| `omniruta --puerto 3000` | Establezca el puerto canónico/API en 3000 | -| `omniruta --mcp` | Inicie el servidor MCP (transporte stdio) | -| `omniroute --no-abierto` | No abrir automáticamente el navegador | -| `omniroute --ayuda` | Mostrar ayuda | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Modo de puerto dividido opcional:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -Para la mayoría de las implementaciones, solo necesita: +For most deployments, you only need: -| Variables | Predeterminado | Propósito | -| ------------------------ | ----------------------- | --------------------------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600000` | Línea de base compartida para recuperación ascendente, tiempos de espera de Undici ocultos, solicitudes de huellas digitales TLS y tiempos de espera de proxy/solicitud de puente API | -| `STREAM_IDLE_TIMEOUT_MS` | hereda `REQUEST_TIMEOUT_MS` | Brecha máxima entre fragmentos de transmisión antes de que OmniRoute cancele la transmisión SSE | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -Se conserva la compatibilidad con versiones anteriores: `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` y otras variables de tiempo de espera por capa aún funcionan y anulan la línea base compartida. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Las anulaciones avanzadas están disponibles si necesita un control más preciso:| Variables | Predeterminado | Propósito | -| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | hereda `REQUEST_TIMEOUT_MS` | Tiempo de espera total de solicitudes ascendentes utilizado por la señal de aborto de recuperación principal | -| `FETCH_HEADERS_TIMEOUT_MS` | hereda `FETCH_TIMEOUT_MS` | Límite de tiempo de Undici para recibir encabezados de respuesta ascendentes | -| `FETCH_BODY_TIMEOUT_MS` | hereda `FETCH_TIMEOUT_MS` | Límite de tiempo undici entre fragmentos de cuerpo ascendentes (`0` lo desactiva) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Tiempo de espera de conexión TCP de Undici | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici tiempo de espera del socket de mantenimiento activo inactivo | -| `TLS_CLIENT_TIMEOUT_MS` | hereda `FETCH_TIMEOUT_MS` | Tiempo de espera para solicitudes de huellas digitales TLS realizadas a través de `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | hereda `REQUEST_TIMEOUT_MS` o `30000` | Tiempo de espera para el reenvío de proxy `/v1` desde el puerto API al puerto del panel | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Tiempo de espera de solicitud entrante en el servidor puente API | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Tiempo de espera del encabezado entrante en el servidor puente API | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Tiempo de espera de mantenimiento de actividad en el servidor puente API | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Tiempo de espera de inactividad del socket en el servidor puente API (`0` lo deshabilita) | +Advanced overrides are available if you need finer control: -Si ejecuta OmniRoute detrás de Nginx, Caddy, Cloudflare u otro proxy inverso, asegúrese de que el proxy -Los tiempos de espera también son mayores que los tiempos de espera de transmisión/recuperación de OmniRoute.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Abra Panel → `Proveedores` y conecte al menos un proveedor (clave OAuth o API). -2. Abra Panel → `Endpoints` y cree una clave API. -3. (Opcional) Abra el Panel → `Combos` y configure su cadena alternativa.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Funciona con Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode y SDK compatibles con OpenAI.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (para operaciones basadas en herramientas):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` +Then connect your MCP client over `stdio` and test tools like: -Luego conecte su cliente MCP a través de `stdio` y pruebe herramientas como: +- `omniroute_get_health` +- `omniroute_list_combos` --`omniroute_get_health` --`omniroute_list_combos` +**A2A (for agent-to-agent workflows):** -**A2A (para flujos de trabajo de agente a agente):**```bash +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -759,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Esta suite valida flujos de clientes MCP y A2A reales frente a una aplicación en ejecución.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -767,13 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` - -Void Linux (plantilla `xbps-src`) +
+Void Linux (`xbps-src` template) -Para los usuarios de Void Linux, pueden crear un paquete nativo usando `xbps-src`. Guarde este bloque como `srcpkgs/omniroute/template`:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -785,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -793,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -869,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -880,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute está disponible como imagen pública de Docker en [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Ejecución rápida:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -890,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**Con archivo de entorno:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Usando Docker Compose:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -El soporte del panel para implementaciones de Docker ahora incluye un**Cloudflare Quick Tunnel**con un solo clic en "Panel → Endpoints". La primera habilitación descarga `cloudflared` solo cuando es necesario, inicia un túnel temporal hacia su punto final `/v1` actual y muestra la URL `https://*.trycloudflare.com/v1` generada directamente debajo de su URL pública normal. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Notas: +Notes: -- Las URL de Quick Tunnel son temporales y cambian después de cada reinicio. -- Los túneles rápidos no se restauran automáticamente después de reiniciar OmniRoute o un contenedor. Vuelva a habilitarlos desde el panel cuando sea necesario. -- La instalación administrada actualmente es compatible con Linux, macOS y Windows en `x64`/`arm64`. -- Los túneles rápidos administrados utilizan de forma predeterminada el transporte HTTP/2 para evitar ruidosas advertencias de búfer QUIC UDP en entornos de contenedores restringidos. Configure `CLOUDFLARED_PROTOCOL=quic` o `auto` si desea un transporte diferente. -- Las imágenes de Docker agrupan las raíces de CA del sistema y las pasan a "cloudflared" administrado, lo que evita fallas de confianza de TLS cuando el túnel se inicia dentro del contenedor. -- SQLite se ejecuta en modo WAL. Se debe permitir que `docker stop` finalice para que OmniRoute pueda verificar los últimos cambios en `storage.sqlite`. -- Los archivos Compose incluidos ya establecen un período de gracia de parada de 40 segundos. Si ejecuta la imagen directamente, mantenga `--stop-timeout 40` (o similar) para que las paradas manuales no interrumpan la limpieza del apagado. -- Configure `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` si desea que OmniRoute use un binario existente en lugar de descargar uno. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Usando Docker Compose con Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute se puede exponer de forma segura mediante el aprovisionamiento SSL automático de Caddy. Asegúrese de que el registro DNS A de su dominio apunte a la IP de su servidor.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` - -| Imagen | Etiqueta | Tamaño | Descripción | +| Image | Tag | Size | Description | | ------------------------ | -------- | ------ | --------------------- | -| `diegosouzapw/omniroute` | `último` | ~250MB | Última versión estable | -| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Versión actual |--- +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | + +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**¡NUEVO!**OmniRoute ahora está disponible como**aplicación de escritorio nativa**para Windows, macOS y Linux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Ejecute OmniRoute como una aplicación de escritorio independiente: no se requiere terminal, navegador ni Internet para los modelos locales. La aplicación basada en Electron incluye: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Ventana nativa**: ventana de aplicación dedicada con integración de la bandeja del sistema -- 🔄**Inicio automático**: inicie OmniRoute al iniciar sesión en el sistema -- 🔔**Notificaciones nativas**: reciba alertas sobre el agotamiento de la cuota o problemas con el proveedor -- ⚡**Instalación con un clic**: NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Modo sin conexión**: funciona completamente sin conexión con el servidor incluido### Inicio Rápido +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Inicio Rápido ```bash # Development mode @@ -979,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -Cuando está minimizado, OmniRoute reside en la bandeja del sistema con acciones rápidas: +When minimized, OmniRoute lives in your system tray with quick actions: -- Abrir panel -- Cambiar puerto del servidor -- Salir de la aplicación +- Open dashboard +- Change server port +- Quit application -📖 Documentación completa: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Nivel | Proveedor | Costo | Restablecer cuota | Mejor para | -| ------------------ | --------------------------------------- | -------------------------------------- | ----------------------------- | ---------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 SUSCRIPCIÓN** | Código Claude (Pro) | $20/mes | 5h + semanales | Ya suscrito | -| | Códice (Plus/Pro) | $20-200/mes | 5h + semanales | Usuarios de OpenAI | -| | Géminis CLI | **GRATIS** | 180K/mes + 1K/día | ¡Todos! | -| | Copiloto de GitHub | $10-19/mes | Mensual | Usuarios de GitHub | -| **🔑 CLAVE API** | NIM de NVIDIA | **GRATIS**(desarrollador para siempre) | ~40 RPM | Más de 70 modelos abiertos | -| | Cerebras | **GRATIS**(1 millón de tok/día) | 60.000 TPM / 30 RPM | El más rápido del mundo | -| | Groq | **GRATIS**(30 RPM) | 14,4K RPD | Llama/Gemma ultrarrápida | -| | DeepSeek V3.2 | $0,27/$1,10 por 1 millón | Ninguno | Mejor razonamiento precio/calidad | -| | xAI Grok-4 Rápido | **$0,20/$0,50 por 1M**🆕 | Ninguno | Llamada de herramienta + más rápida, ultrabaja | -| | xAI Grok-4 (estándar) | $0,20/$1,50 por 1 millón 🆕 | Ninguno | Insignia de razonamiento de xAI | -| | Mistral | Prueba gratuita + pago | Tarifa limitada | IA europea | -| | Enrutador abierto | Pago por uso | Ninguno | Más de 100 modelos agregados. | -| **💰 BARATO** | GLM-5 (vía Z.AI) 🆕 | 0,5 dólares/1 millón | Todos los días a las 10 a. m. | Salida de 128K, el buque insignia más nuevo | -| | GLM-4.7 | 0,6 dólares/1 millón | Todos los días a las 10 a. m. | Respaldo presupuestario | -| | MiniMax M2.5 🆕 | 0,3 $/1 millón de entrada | 5 horas rodantes | Razonamiento + tareas agentes | -| | MiniMax M2.1 | 0,2 dólares/1 millón | 5 horas rodantes | Opción más barata | -| | Kimi K2.5 (API Moonshot) 🆕 | Pago por uso | Ninguno | Acceso directo a la API Moonshot | -| | Kimi K2 | $9/mes fijo | 10 millones de tokens/mes | Costo predecible | -| **🆓 GRATIS** | Qoder | **$0** | Ilimitado | 5 modelos ilimitados | -| | Qwen | **$0** | Ilimitado | 4 modelos ilimitados | -| | kiro | **$0** | Ilimitado | Claude Sonnet/Haiku (constructor de AWS) | -| | LongCat Flash Lite 🆕 | **$0**(50 millones de tok/día 🔥) | 1 RPS | La cuota gratuita más grande del mundo | -| | Polinizaciones AI 🆕 | **$0**(no se necesita clave) | 1 solicitud/15 s | GPT-5, Claude, DeepSeek, Llama 4 | -| | IA de los trabajadores de Cloudflare 🆕 | **$0**(10K Neuronas/día) | ~150 resp/día | Más de 50 modelos, ventaja global | -| | Escala de IA 🆕 | **$0**(1 millón de tokens en total) | Tarifa limitada | UE/RGPD, Qwen3 235B, Llama 70B | > 🆕**Nuevos modelos agregados (marzo de 2026):**Familia Grok-4 Fast a $0,20/$0,50/M (comparado a 1143 ms: 30 % más rápido que Gemini 2.5 Flash), GLM-5 a través de Z.AI con salida de 128 K, razonamiento MiniMax M2.5, precios actualizados de DeepSeek V3.2, Kimi K2.5 a través de API directa Moonshot. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 Pila combinada de $0: la configuración gratuita completa:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Costo cero. Nunca deja de codificar.**Configure esto como un combo OmniRoute y todos los respaldos se realizarán automáticamente, sin cambios manuales.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Todos los modelos a continuación son**100% gratuitos y no se requiere tarjeta de crédito**. OmniRoute realiza rutas automáticas entre ellos cuando se agota una cuota; combínelos todos para obtener una combinación irrompible de $0.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Modelo | Prefijo | Límite | Límite de tarifa | +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | --------------------- | -| `claude-soneto-4.5` | `kr/` |**Ilimitado**| No se ha informado de un límite diario | -| `claude-haiku-4.5` | `kr/` |**Ilimitado**| No se ha informado de un límite diario | -| `claude-opus-4.6` | `kr/` |**Ilimitado**| Última obra a través de Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -| Modelo | Prefijo | Límite | Límite de tarifa | +### 🟢 QODER MODELS (Free PAT via qodercli) + +| Model | Prefix | Limit | Rate Limit | | ------------------ | ------ | ------------- | --------------- | -| `kimi-k2-pensamiento` | `si/` |**Ilimitado**| No hay límite reportado | -| `qwen3-codificador-plus` | `si/` |**Ilimitado**| No hay límite reportado | -| `deepseek-r1` | `si/` |**Ilimitado**| No hay límite reportado | -| `minimax-m2.1` | `si/` |**Ilimitado**| No hay límite reportado | -| `kimi-k2` | `si/` |**Ilimitado**| No hay límite reportado | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -> Método de conexión recomendado:**Token de acceso personal + `qodercli`**. El navegador OAuth es -> experimental y deshabilitado de forma predeterminada a menos que las variables de entorno `QODER_OAUTH_*` estén configuradas.### 🟡 QWEN MODELS (Device Code Auth) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| Modelo | Prefijo | Límite | Límite de tarifa | +### 🟡 QWEN MODELS (Device Code Auth) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | ------------------- | -| `qwen3-codificador-plus` | `qw/` |**Ilimitado**| No hay límite reportado | -| `qwen3-codificador-flash` | `qw/` |**Ilimitado**| No hay límite reportado | -| `qwen3-codificador-siguiente` | `qw/` |**Ilimitado**| No hay límite reportado | -| `modelo-visión` | `qw/` |**Ilimitado**| Multimodal (imágenes) |### 🟣 GEMINI CLI (Google OAuth) +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Modelo | Prefijo | Límite | Límite de tarifa | +### 🟣 GEMINI CLI (Google OAuth) + +| Model | Prefix | Limit | Rate Limit | | ------------------------ | ------ | --------------------------- | ------------- | -| `gemini-3-flash-preview` | `gc/` |**180.000 tok/mes**+ 1.000/día | Reinicio mensual | -| `géminis-2.5-pro` | `gc/` | 180K/mes (piscina compartida) | Alta calidad |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -| Nivel | Límite diario | Límite de tarifa | Notas | +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---------- | ------------ | ----------- | ------------------------------------------------------ | -| Gratis (desarrollador) | Sin límite de fichas |**~40 RPM**| Más de 70 modelos; transición a límites de tasa pura a mediados de 2025 | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -Modelos gratuitos populares: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -| Nivel | Límite diario | Límite de tarifa | Notas | -| ---- | ----------------- | ---------------- | ------------------------------------- | -| Gratis |**1 millón de tokens/día**| 60.000 TPM / 30 RPM | La inferencia LLM más rápida del mundo; se reinicia diariamente | +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -Disponible gratis: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -| Nivel | Límite diario | Límite de tarifa | Notas | +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` + +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---- | ------------- | ---------------- | ----------------------------------------- | -| Gratis |**14,4K RPD**| 30 RPM por modelo | Sin tarjeta de crédito; 429 en límite, sin cargo | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -Disponible gratis: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -| Modelo | Prefijo | Cuota Diaria Gratuita | Notas | -| ----------------------- | ------ | ----------------- | ----------------------- | -| `LongCat-Flash-Lite` | `lc/` |**50 millones de tokens**💥 | La cuota gratuita más grande de la historia | -| `LongCat-Flash-Chat` | `lc/` | Fichas de 500.000 | Chat multiturno | -| `LongCat-Flash-Pensamiento` | `lc/` | Fichas de 500.000 | Razonamiento / CoT | -| `LongCat-Flash-Thinking-2601` | `lc/` | Fichas de 500.000 | Versión de enero de 2026 | -| `LongCat-Flash-Omni-2603` | `lc/` | Fichas de 500.000 | Multimodal | +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 -> 100% gratis mientras estés en la versión beta pública. Regístrese en [longcat.chat](https://longcat.chat) con correo electrónico o teléfono. Se reinicia diariamente a las 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | -| Modelo | Prefijo | Límite de tarifa | Proveedor detrás | +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. + +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| `openai` | `pol/` | 1 solicitud/15 s | GPT-5 | -| `claude` | `pol/` | 1 solicitud/15 s | Claude antrópico | -| `géminis` | `pol/` | 1 solicitud/15 s | Google Géminis | -| `búsqueda profunda` | `pol/` | 1 solicitud/15 s | Búsqueda profunda V3 | -| `llama` | `pol/` | 1 solicitud/15 s | Meta Llama 4 Explorador | -| `mistral` | `pol/` | 1 solicitud/15 s | Mistral IA | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**Cero fricción:**Sin registro, sin clave API. Agregue el proveedor de polinizaciones con un campo clave vacío y funcionará de inmediato.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| Nivel | Neuronas Diarias | Uso equivalente | Notas | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 + +| Tier | Daily Neurons | Equivalent Usage | Notes | | ---- | ------------- | --------------------------------------- | ----------------------- | -| Gratis |**10.000**| ~150 LLM resp / audio 500s / 15K incrustaciones | Ventaja global, más de 50 modelos | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -Modelos gratuitos populares: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (¡audio gratis!), `@cf/qwen/qwen2.5-coder-15b-instruct` +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -> Requiere token API + ID de cuenta de [dash.cloudflare.com](https://dash.cloudflare.com). Almacene la identificación de la cuenta en la configuración del proveedor.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -| Nivel | Cuota Gratuita | Ubicación | Notas | +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | | ---- | ------------- | ------------ | ----------------------------------- | -| Gratis |**1 millón de tokens**| 🇫🇷 París, UE | No se necesita tarjeta de crédito dentro de los límites | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | -Disponible gratis: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` -> Cumple con la UE/GDPR. Obtenga la clave API en [console.scaleway.com](https://console.scaleway.com). +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). ->**💡 El paquete gratuito definitivo (11 proveedores, $0 para siempre):** +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Soneto/Haiku ILIMITADO -> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 ILIMITADO -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 millones de tokens/día 🔥 -> Polinizaciones (pol/) → GPT-5, Claude, DeepSeek, Llama 4: no se necesita clave -> Qwen (qw/) → modelos de codificador qwen3 ILIMITADOS -> Gemini (gemini/) → Gemini 2.5 Flash: 1.500 solicitudes/día gratis -> Cloudflare AI (cf/) → Más de 50 modelos: 10.000 neuronas/día -> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1 millón de tokens gratis (UE) -> Groq (groq/) → Llama/Gemma — 14,4K solicitudes/día ultrarrápidas -> NVIDIA NIM (nvidia/) → Más de 70 modelos abiertos: 40 RPM para siempre -> Cerebras (cerebras/) → Llama/Qwen más rápido del mundo: 1 millón de tok/día -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Transcribe cualquier audio/video por**$0**: Deepgram ofrece $200 gratis, un respaldo de $50 para AssemblyAI y Groq Whisper como respaldo de emergencia ilimitado. +## 🎙️ Free Transcription Combo -| Proveedor | Créditos gratis | Mejor modelo | Límite de tarifa | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. + +| Provider | Free Credits | Best Model | Rate Limit | | ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | -| 🟢**Deepgrama**|**$200 gratis**(registro) | `nova-3`: máxima precisión, más de 30 idiomas | Sin límite de RPM en créditos gratis | -| 🔵**AsambleaAI**|**$50 gratis**(registro) | `universal-3-pro` — capítulos, sentimiento, PII | Sin límite de RPM en créditos gratis | -| 🔴**Groq**|**Gratis para siempre**| `whisper-large-v3` — OpenAI Whisper | 30 RPM (velocidad limitada) | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | -**Combo sugerido en `/dashboard/combos`:**``` +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Luego, en `/dashboard/media` → pestaña**Transcripción**: cargue cualquier archivo de audio o video → seleccione su punto final combinado → obtenga la transcripción en formatos compatibles.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 está diseñado como una plataforma operativa, no solo como un proxy de retransmisión.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Característica | Qué hace | -| ----------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Familia rápida Grok-4** | Modelos xAI a $0,20/$0,50/M: comparado con 1143 ms (30% más rápido que Gemini 2.5 Flash) | -| 🧠**GLM-5 vía Z.AI** | Contexto de salida de 128.000 dólares, 0,5 dólares/1 millón: el buque insignia más nuevo de la familia GLM | -| 🔮**MiniMax M2.5** | Razonamiento + tareas de agente a 0,30 USD/1 millón: mejora significativa desde M2.1 | -| 🎯**marcador de llamadas de herramientas por modelo** | `toolCalling: verdadero/falso` por modelo en el registro: AutoCombo omite los modelos que no son compatibles con herramientas | -| 🌍**Detección de intención multilingüe** | Palabras clave PT/ZH/ES/AR en la puntuación AutoCombo: mejor selección de modelos para contenido que no está en inglés | -| 📊**Retrocesos impulsados ​​por los índices de referencia** | Latencia p95 real de solicitudes en vivo alimenta puntuación combinada: AutoCombo aprende de datos reales | -| 🔁**Solicitar deduplicación** | Ventana de deduplicación basada en hash de contenido: segura para múltiples agentes, evita cargos duplicados | -| 🔌**Estrategia de enrutador conectable** | Interfaz extensible `RouterStrategy`: agregue lógica de enrutamiento personalizada como complementos | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Característica | Qué hace | -| ------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Patio de juegos modelo** | Página de panel para probar cualquier modelo directamente: selectores de proveedor/modelo/punto final, editor Monaco, transmisión, cancelación, sincronización | -| 🔏**Coincidencia de huellas dactilares CLI** | Orden de encabezado/cuerpo por proveedor para que coincida con las firmas CLI nativas: alterne por proveedor en Configuración > Seguridad.**Se conserva la IP de tu proxy** | -| 🤝**Soporte ACP (Protocolo cliente-agente)** | Descubrimiento de agentes CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw y 9 más), generador de procesos, punto final `/api/acp/agents` | -| 🤖**Panel de agentes de ACP** | Página Depurar › Agentes: cuadrícula de 14 agentes con estado de instalación, versión y formulario de agente personalizado para cualquier herramienta CLI. Los usuarios de**OpenCode**obtienen un botón "Descargar opencode.json" que genera automáticamente una configuración lista para usar con todos los modelos disponibles. | -| 🔧**Enrutamiento del modelo personalizado `apiFormat`** | Los modelos personalizados con `apiFormat: "responses"` ahora se enrutan correctamente al traductor de la API de Respuestas | -| 🏢**Aislamiento del espacio de trabajo del Codex** | Múltiples espacios de trabajo de Codex por correo electrónico: OAuth separa correctamente las conexiones por ID del espacio de trabajo | -| 🔄**Actualización automática electrónica** | La aplicación de escritorio busca actualizaciones + instalación automática al reiniciar | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Característica | Qué hace | -| --------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**Servidor MCP (25 herramientas)** | Herramientas IDE/agente a través de 3 transportes: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 núcleos + 3 memorias + 4 herramientas de habilidades | -| 🤝**Servidor A2A (JSON-RPC + SSE)** | Ejecución de tareas de agente a agente con flujos de sincronización y streaming | -| 🧭**Página de puntos finales consolidados** | Página de administración con pestañas con pestañas Endpoint Proxy, MCP, A2A y API Endpoints | -| 🎚️**Activación/desactivación de servicio** | Interruptores ON/OFF para MCP y A2A con persistencia de configuración (predeterminado: OFF) | -| 🛰️**Latido del tiempo de ejecución de MCP** | Estado real del proceso (pid, tiempo de actividad, antigüedad del latido, transporte, modo de alcance) | -| 📋**Pista de auditoría de MCP** | Registros de auditoría filtrables con éxito/fracaso y atribución de claves | -| 🔐**Cumplimiento del alcance del MCP** | 10 permisos de alcance granular para acceso controlado a herramientas | -| 📡**Gestión del ciclo de vida de tareas A2A** | Enumerar/filtrar tareas, inspeccionar eventos/artefactos, cancelar tareas en ejecución | -| 📋**Descubrimiento de tarjeta de agente** | `/.well-known/agent.json` para el descubrimiento automático de clientes | -| 🧪**Arnés de prueba del protocolo E2E** | El cliente real MCP SDK + A2A fluye en `test:protocols:e2e` | -| ⚙️**Controles operativos** | Cambie el combo, aplique perfiles de resiliencia, reinicie los disyuntores desde una superficie de control | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Característica | Qué hace | -| ----------------------------------------------- | ----------------------------------------------------------------------------------------------- | ----------------------- | -| 🎯**Retroceso inteligente de 4 niveles** | Ruta automática: Suscripción → Clave API → Barato → Gratis | -| 📊**Seguimiento de cuotas en tiempo real** | Recuento de tokens en vivo + reinicio de cuenta regresiva por proveedor | -| 🔄**Traducción de formato** | OpenAI ↔ Claude ↔ Gemini ↔ Respuestas con conversiones seguras para esquemas | -| 👥**Soporte multicuenta** | Múltiples cuentas por proveedor con selección inteligente | -| 🔄**Actualización automática de tokens** | Los tokens de OAuth se actualizan automáticamente con un reintento | -| 🎨**Combinaciones personalizadas** | 9 estrategias de equilibrio + control de la cadena alternativa | -| 🌐**Enrutador comodín** | `proveedor/*` enrutamiento dinámico | -| 🧠**Pensando en los controles presupuestarios** | Límites de razonamiento de transferencia, automático, personalizado y adaptativo | -| 🔀**Alias ​​de modelo** | Seguridad de migración y alias de modelo integrado y personalizado | -| ⚡**Degradación del fondo** | Dirija tareas en segundo plano de baja prioridad a modelos más baratos | -| 🧪**Enrutamiento inteligente basado en tareas** | Modelo de selección automática por tipo de contenido (codificación/visión/análisis/resumen) | -| 🔄**Flujos de trabajo del agente A2A** | Orquestador FSM determinista para ejecuciones de agentes de varios pasos con estado | -| 🔀**Enrutamiento adaptativo** | Anulación de estrategia dinámica basada en el volumen de tokens y la complejidad del aviso | -| 🎲**Diversidad de proveedores** | Puntuación de entropía de Shannon que equilibra la distribución del tráfico de combo automático | -| 💬**Inyección de indicación del sistema** | Controles de comportamiento global aplicados consistentemente | -| 📄**Compatibilidad API de respuestas** | Soporte completo `/v1/responses` para Codex y flujos de trabajo agentes avanzados | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Característica | Qué hace | -| ---------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**Generación de imágenes** | `/v1/images/generaciones` con backends locales y en la nube | -| 📐**Incrustaciones** | `/v1/embeddings` para búsqueda y canales RAG | -| 🎤**Transcripción de audio** | `/v1/audio/transcriptions` — 7 proveedores (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), detección automática de idioma, compatibilidad con MP4/MP3/WAV | -| 🔊**Texto a voz** | `/v1/audio/speech` — 10 proveedores (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) con mensajes de error correctos | -| 🎬**Generación de vídeo** | `/v1/videos/generaciones` (flujos de trabajo ComfyUI + SD WebUI) | -| 🎵**Generación Musical** | `/v1/music/generaciones` (flujos de trabajo de ComfyUI) | -| 🛡️**Moderaciones** | `/v1/moderaciones` controles de seguridad | -| 🔀**Reclasificación** | `/v1/rerank` para puntuación de relevancia | -| 🔍**Búsqueda web**🆕 | `/v1/search` — 5 proveedores (Serper, Brave, Perplexity, Exa, Tavily), más de 6500 gratis/mes, conmutación por error automática, caché | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Característica | Qué hace | -| ---------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**Disyuntores** | Viaje/recuperación por modelo con controles de umbral | -| 🎯**Modelos compatibles con endpoints** | Los modelos personalizados declaran puntos finales compatibles + formato API | -| 🛡️**Rebaño Anti-Truenos** | Mutex + protecciones de semáforo en eventos de reintento/tasa | -| 🧠**Caché semántico + firma** | Reducción de costos/latencia con dos capas de caché | -| ⚡**Solicitar Idempotencia** | Ventana de protección duplicada | -| 🔒**Suplantación de huellas dactilares TLS** | Huella digital TLS similar a la de un navegador:**reduce la detección de bots y el marcado de cuentas** | -| 🔏**Coincidencia de huellas dactilares CLI** | Coincide con las firmas de solicitudes CLI nativas:**reduce el riesgo de prohibición y al mismo tiempo preserva la IP del proxy** | -| 🌐**Filtrado de IP** | Control de lista blanca/lista negra para implementaciones expuestas | -| 📊**Límites de tarifas editables** | Límites globales/a nivel de proveedor configurables con persistencia | -| 📉**Degradación elegante** | Respaldos de capacidad multicapa que protegen las operaciones centrales de la puerta de enlace | -| 📜**Pista de auditoría de configuración** | Seguimiento de cambios basado en diferencias que evita la deriva operativa con reversiones simples | -| ⏳**Sincronización de salud del proveedor** | Monitoreo proactivo de vencimiento de tokens que activa alertas antes de fallas de autorización | -| 🚪**Desactivación automática de cuentas prohibidas** | Disyuntor operativo que sella automáticamente cuentas simbólicas bloqueadas permanentemente | -| 🔑**Administración de claves API + Alcance** | Emisión/rotación de claves segura y controles de modelo/proveedor | -| 👁️**Revelación de clave API con alcance**🆕 | Recuperación voluntaria de claves API a través de `ALLOW_API_KEY_REVEAL` | -| 🛡️**Protegido `/modelos`** | Puerta de autenticación opcional y ocultación de proveedores para el catálogo de modelos | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Característica | Qué hace | -| ----------------------------------------- | ---------------------------------------------------------------------------------- | ---------------------------- | -| 📝**Solicitud + Registro de proxy** | Solicitud/respuesta completa y registro de proxy | -| 📉**Registros detallados transmitidos**🆕 | Reconstruye secuencias de carga útil SSE limpiamente en la interfaz de usuario | -| 📋**Panel de registros unificado** | Vistas de solicitud, proxy, auditoría y consola en una sola página | -| 🔍**Solicitar telemetría** | Latencia p50/p95/p99 y seguimiento de solicitudes | -| 🏥**Panel de salud** | Tiempo de actividad, estados de los interruptores, bloqueos, estadísticas de caché | -| 💰**Seguimiento de costos** | Controles de presupuesto y visibilidad de precios por modelo | -| 📈**Visualizaciones analíticas** | Información sobre el uso de modelos/proveedores y vistas de tendencias | -| 🧪**Marco de evaluación** | Prueba de set dorado con estrategias de partido configurables | -| 📡**Diagnóstico en vivo**🆕 | Omisión de caché semántica para pruebas combinadas en vivo precisas | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Característica | Qué hace | -| --------------------------------------- | ------------------------------------------------------------------------------------- | --------------------- | -| 🌐**Implementar en cualquier lugar** | Localhost, VPS, Docker, entornos Cloud | -| 🚇**Túnel Cloudflare**🆕 | Integración de Quick Tunnel con un clic desde el panel | -| 🔑**Filtrado de modelo de clave API** | Respuesta nativa /v1/models filtrada mediante roles de contexto de portador asignados | -| ⚡**Omisión de caché inteligente** | Heurísticas TTL configurables y controles de recuperación forzada | -| 🔄**Copia de seguridad/Restaurar** | Flujos de exportación/importación y recuperación ante desastres | -| 🧙**Asistente de incorporación** | Configuración guiada de primera ejecución | -| 🔧**Panel de herramientas CLI** | Configuración con un clic para herramientas de codificación populares | -| 🎮**Patio de juegos modelo** | Pruebe cualquier proveedor/modelo/punto final desde el panel | -| 🔏**Alternar huella digital CLI** | Coincidencia de huellas dactilares por proveedor en Configuración > Seguridad | -| 🌐**i18n (30 idiomas)** | Panel completo + compatibilidad con idiomas de documentos con cobertura RTL | -| 🧹**Borrar todos los modelos** | Borrado de la lista de modelos con un solo clic en los detalles del proveedor | -| 👁️**Controles de la barra lateral**🆕 | Ocultar componentes e integraciones desde Configuración de apariencia | -| 📋**Plantillas de problemas** | Plantillas de GitHub estandarizadas para errores y funciones | -| 📂**Directorio de datos personalizado** | Anulación de `DATA_DIR` para la ubicación de almacenamiento | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1292,103 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -Cuando falla la cuota, la tasa o el estado, OmniRoute pasa automáticamente al siguiente candidato sin necesidad de cambiar manualmente.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A se pueden descubrir en la interfaz de usuario y en los documentos (no están ocultos) -- Las API de estado del protocolo exponen datos operativos en vivo (`/api/mcp/*`, `/api/a2a/*`) -- Los paneles incluyen acciones para las operaciones del día 2 (cambio de combo, reinicio de interruptores, cancelación de tareas)#### Translator + validation workflow +#### Protocol management that is visible and operable -El área de Traductor incluye: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Parque infantil**: solicitar comprobaciones de transformación -**Chat Tester**: solicitud/respuesta completa de ida y vuelta -**Banco de pruebas**: varios casos en una ejecución -**Live Monitor**: vista del tráfico en tiempo real +#### Translator + validation workflow -Además de validación de protocolo con clientes reales a través de `npm run test:protocols:e2e`. +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Referencia de herramientas, configuraciones IDE y ejemplos de clientes +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[README del servidor A2A](src/lib/a2a/README.md)**— Habilidades, métodos JSON-RPC, transmisión y ciclo de vida de las tareas## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute incluye un marco de evaluación integrado para probar la calidad de la respuesta de LLM frente a un conjunto de referencia. Acceda a él a través de**Análisis → Evaluaciones**en el panel.### Built-in Golden Set +## 🧪 Evaluations (Evals) -El "OmniRoute Golden Set" precargado contiene casos de prueba para: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Saludos, matemáticas, geografía, generación de código. -- Cumplimiento del formato JSON, traducción, generación de rebajas. -- Rechazo de seguridad (contenido nocivo), conteo, lógica booleana### Evaluation Strategies +### Built-in Golden Set -| Estrategia | Descripción | Ejemplo | -| ------------------- | ---------------------------------------------------------------------------------- | ---------------------------------- | --- | -| `exacto` | La salida debe coincidir exactamente | `"4"` | -| `contiene` | La salida debe contener una subcadena (no distingue entre mayúsculas y minúsculas) | `"París"` | -| `expresión regular` | La salida debe coincidir con el patrón de expresiones regulares | `"1.*2.*3"` | -| `personalizado` | La función JS personalizada devuelve verdadero/falso | `(salida) => salida.longitud > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) - -🧩 Configuración de MCP (Protocolo de contexto del modelo) +
+🧩 MCP Setup (Model Context Protocol) -Inicie el transporte MCP en modo stdio:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Flujo de validación recomendado: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Conecte su cliente MCP a través de stdio. -2. Ejecute `omniroute_get_health`. -3. Ejecute `omniroute_list_combos`. -4. Abra `/dashboard/mcp` para confirmar el latido, la actividad y la auditoría. +Useful APIs for automation: -API útiles para la automatización: +- `GET /api/mcp/status` +- `GET /api/mcp/tools` +- `GET /api/mcp/audit` +- `GET /api/mcp/audit/stats` -- `OBTENER /api/mcp/status` -- `OBTENER /api/mcp/tools` -- `OBTENER /api/mcp/auditoría` -- `OBTENER /api/mcp/audit/stats`
+ - -🤝 Configuración de A2A (Agent2Agent) +
+🤝 A2A Setup (Agent2Agent) -Descubra el agente:```bash +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Enviar una tarea:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` +Manage lifecycle: -Gestionar el ciclo de vida: +- `GET /api/a2a/status` +- `GET /api/a2a/tasks` +- `GET /api/a2a/tasks/:id` +- `POST /api/a2a/tasks/:id/cancel` -- `OBTENER /api/a2a/status` -- `OBTENER /api/a2a/tareas` -- `OBTENER /api/a2a/tasks/:id` -- `POST /api/a2a/tasks/:id/cancelar` +Operational UI: -Interfaz de usuario operativa: +- `/dashboard/a2a` for task/state/stream observability and smoke actions -- `/dashboard/a2a` para observabilidad de tarea/estado/corriente y acciones de humo
+ - -🧪 Validación de protocolo de un extremo a otro +
+🧪 End-to-end protocol validation -Validar ambos protocolos con clientes reales:```bash +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Esto verifica: +This verifies: -- Conexión/lista/llamada del cliente MCP SDK -- Descubrimiento A2A/enviar/transmitir/obtener/cancelar -- Verificación cruzada de datos en auditoría MCP y API de administración de tareas A2A
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs - -💳 Proveedores de suscripción### Claude Code (Pro/Max) + + +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1401,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Consejo profesional:**Utilice Opus para tareas complejas y Sonnet para mayor velocidad. ¡OmniRoute realiza un seguimiento de la cuota por modelo!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1415,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Cada cuenta de Codex ahora tiene políticas para alternar en `Panel -> Proveedores`: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (ON/OFF): aplica la política de umbral de ventana de 5 horas. -- `Semanal` (ON/OFF): aplica la política de umbral de ventana semanal. -- Comportamiento de umbral: cuando una ventana habilitada alcanza >=90% de uso, esa cuenta se omite. -- Comportamiento de rotación: OmniRoute dirige automáticamente a la siguiente cuenta elegible del Codex. -- Comportamiento de reinicio: cuando pasa el tiempo `resetAt` del proveedor, la cuenta vuelve a ser elegible automáticamente. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Escenarios: +Scenarios: -- `5h ON` + `Weekly ON`: la cuenta se omite cuando cualquiera de las ventanas alcanza el umbral. -- `5h OFF` + `Weekly ON`: solo el uso semanal puede bloquear la cuenta. -- `5h ON` + `Weekly OFF`: solo el uso de 5 horas puede bloquear la cuenta. -- `resetAt` pasó: la cuenta vuelve a ingresar a la rotación automáticamente (no se puede volver a habilitar manualmente).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1440,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Mejor valor:**¡Enorme nivel gratuito! Utilice esto antes de los niveles pagos.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1455,71 +1662,91 @@ Models:
- -🔑 Proveedores de claves API### NVIDIA NIM (FREE developer access — 70+ models) +
+🔑 API Key Providers -1. Regístrate: [build.nvidia.com](https://build.nvidia.com) -2. Obtenga una clave API gratuita (1000 créditos de inferencia incluidos) -3. Panel de control → Agregar proveedor → NVIDIA NIM: - - Clave API: `nvapi-tu-clave` +### NVIDIA NIM (FREE developer access — 70+ models) -**Modelos:**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` y más de 50 +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Consejo profesional:**API compatible con OpenAI: ¡funciona perfectamente con la traducción de formatos de OmniRoute!### DeepSeek +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -1. Regístrate: [platform.deepseek.com](https://platform.deepseek.com) -2. Obtenga la clave API -3. Panel de control → Agregar proveedor → DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -**Modelos:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +### DeepSeek -1. Regístrese: [console.groq.com](https://console.groq.com) -2. Obtenga la clave API (nivel gratuito incluido) -3. Panel de control → Agregar proveedor → Groq +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -**Modelos:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Consejo profesional:**Inferencia ultrarrápida: ¡lo mejor para codificación en tiempo real!### OpenRouter (100+ Models) +### Groq (Free Tier Available!) -1. Regístrate: [openrouter.ai](https://openrouter.ai) -2. Obtenga la clave API -3. Panel de control → Agregar proveedor → OpenRouter +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -**Modelos:**Acceda a más de 100 modelos de los principales proveedores a través de una única clave API. +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Comportamiento del panel:**Los modelos OpenRouter se administran desde**Modelos disponibles**. La adición manual, la importación y la sincronización automática actualizan la misma lista.
+**Pro Tip:** Ultra-fast inference — best for real-time coding! - -💰 Proveedores baratos (copia de seguridad)### GLM-4.7 (Daily reset, $0.6/1M) +### OpenRouter (100+ Models) -1. Regístrate: [Zhipu AI](https://open.bigmodel.cn/) -2. Obtenga la clave API del plan de codificación -3. Panel de control → Agregar clave API: - - Proveedor: `glm` - - Clave API: `tu-clave` +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -**Uso:**`glm/glm-4.7` +**Models:** Access 100+ models from all major providers through a single API key. -**Consejo profesional:**¡El plan de codificación ofrece una cuota triple a un costo de 1/7! Reiniciar diariamente a las 10:00 a.m.### MiniMax M2.1 (5h reset, $0.20/1M) +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -1. Regístrate: [MiniMax](https://www.minimax.io/) -2. Obtenga la clave API -3. Panel de control → Agregar clave API + -**Uso:**`minimax/MiniMax-M2.1` +
+💰 Cheap Providers (Backup) -**Consejo profesional:**¡La opción más barata para contexto largo (1 millón de tokens)!### Kimi K2 ($9/month flat) +### GLM-4.7 (Daily reset, $0.6/1M) -1. Suscríbete: [Moonshot AI](https://platform.moonshot.ai/) -2. Obtenga la clave API -3. Panel de control → Agregar clave API +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**Uso:**`kimi/kimi-latest` +**Use:** `glm/glm-4.7` -**Consejo profesional:**¡Fijo $9/mes por 10 millones de tokens = $0,90/1 millón de costo efectivo!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. - -🆓 Proveedores GRATUITOS (respaldo de emergencia)### Qoder (5 FREE models via OAuth) +### MiniMax M2.1 (5h reset, $0.20/1M) + +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `minimax/MiniMax-M2.1` + +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1560,8 +1787,10 @@ Models:
- -🎨 Crear combos### Example 1: Maximize Subscription → Cheap Backup +
+🎨 Create Combos + +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1589,8 +1818,10 @@ Cost: $0 forever!
- -🔧 Integración CLI### Cursor IDE +
+🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1601,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Utilice la página**Herramientas CLI**en el panel para realizar la configuración con un solo clic o edite `~/.claude/settings.json` manualmente.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1612,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Opción 1: Panel de control (recomendado):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Opción 2 — Manual:**Editar `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1629,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Nota:**OpenClaw solo funciona con OmniRoute local. Utilice `127.0.0.1` en lugar de `localhost` para evitar problemas de resolución de IPv6.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1643,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Paso 1:**Agregue OmniRoute como proveedor personalizado:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Paso 2:**Crea/edita `opencode.json` en la raíz de tu proyecto:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1669,117 +1909,130 @@ opencode } } } -```` +``` -**Paso 3:**Selecciona el modelo en OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Consejo:**Agregue cualquier modelo disponible en su terminal `/v1/models` de OmniRoute a la sección `modelos`. Utilice el formato `proveedor/modelo-id` desde su panel de OmniRoute.
+ --- ## Solución de Problemas - -Haga clic para expandir la guía de solución de problemas +
+Click to expand troubleshooting guide -**"El modelo de idioma no proporcionó mensajes"** +**"Language model did not provide messages"** -- Cuota de proveedor agotada → Verifique el rastreador de cuotas del panel -- Solución: utilice el combo alternativo o cambie a un nivel más económico +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**Limitación de tasa** +**Rate limiting** -- Cuota de suscripción agotada → Alternativa a GLM/MiniMax -- Agregar combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**El token de OAuth expiró** +**OAuth token expired** -- Actualizado automáticamente por OmniRoute -- Si los problemas persisten: Panel → Proveedor → Volver a conectar +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Altos costos** +**High costs** -- Verifique las estadísticas de uso en Panel → Costos -- Cambiar el modelo principal a GLM/MiniMax -- Utilice el nivel gratuito (Gemini CLI, Qoder) para tareas no críticas +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Los puertos del panel/API están incorrectos** +**Dashboard/API ports are wrong** -- `PORT` es el puerto base canónico (y el puerto API por defecto) -- `API_PORT` anula sólo el detector de API compatible con OpenAI -- `DASHBOARD_PORT` anula solo el panel de control/escucha Next.js -- Configure `NEXT_PUBLIC_BASE_URL` en su panel/URL pública (para devoluciones de llamada de OAuth) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Errores de sincronización en la nube** +**Cloud sync errors** -- Verifique que `BASE_URL` apunte a su instancia en ejecución -- Verifique que `CLOUD_URL` apunte al punto final de nube esperado -- Mantenga los valores `NEXT_PUBLIC_*` alineados con los valores del lado del servidor +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**El primer inicio de sesión no funciona** +**First login not working** -- Marque `INITIAL_PASSWORD` en `.env` -- Si no está configurada, la contraseña alternativa es `123456` +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**No hay registros de solicitudes** +**No request logs** -- Los artefactos de solicitud se escriben en `DATA_DIR/call_logs/` como un archivo JSON por solicitud -- Habilite la captura de canalización desde Panel → Registros → Solicitar registros si necesita cargas útiles detalladas por etapa -- Configure `APP_LOG_TO_FILE=true` si también desea que la consola de la aplicación registre `logs/application/app.log` -- Ajuste `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` y `CALL_LOG_MAX_ENTRIES` según sea necesario +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**La prueba de conexión muestra "No válido" para proveedores compatibles con OpenAI** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Muchos proveedores no exponen un punto final `/models` -- OmniRoute v1.0.6+ incluye validación alternativa mediante la finalización del chat -- Asegúrese de que la URL base incluya el sufijo `/v1`### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix - +### 🔐 OAuth on a Remote Server + + ->**⚠️ Importante para los usuarios que ejecutan OmniRoute en un VPS, Docker o cualquier servidor remoto**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -Los proveedores**Antigravity**y**Gemini CLI**utilizan**Google OAuth 2.0**. Google requiere que `redirect_uri` en el flujo de OAuth coincida exactamente con uno de los URI registrados previamente en Google Cloud Console de la aplicación. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -Las credenciales de OAuth incluidas en OmniRoute están registradas**solo para `localhost`**. Cuando accede a OmniRoute en un servidor remoto (por ejemplo, `https://omniroute.myserver.com`), Google rechaza la autenticación con:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Debes crear un**ID de cliente de OAuth 2.0**en Google Cloud Console con el URI de tu servidor.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Abra la consola de Google Cloud** +#### Step-by-step -Vaya a: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Cree un nuevo ID de cliente OAuth 2.0** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Haga clic en**"+ Crear credenciales"**→**"ID de cliente OAuth"** -- Tipo de aplicación:**"Aplicación web"** -- Nombre: lo que quieras (por ejemplo, `OmniRoute Remote`) +**2. Create a new OAuth 2.0 Client ID** -**3. Agregar URI de redireccionamiento autorizado** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -En el campo**"URI de redireccionamiento autorizado"**, agregue:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Reemplace `your-server.com` con el dominio o IP de su servidor (incluya el puerto si es necesario, por ejemplo, `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. Guarde y copie las credenciales** +After creating, Google will show the **Client ID** and **Client Secret**. -Después de la creación, Google mostrará el**ID de cliente**y el**Secreto de cliente**. +**5. Set environment variables** -**5. Establecer variables de entorno** +In your `.env` (or Docker environment variables): -En su `.env` (o variables de entorno de Docker):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1788,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Reiniciar OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Intente conectarse nuevamente** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Panel → Proveedores → Antigravity (o Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google ahora redirigirá correctamente a `https://your-server.com/callback`.--- +--- #### Temporary workaround (without custom credentials) -Si no desea configurar sus propias credenciales en este momento, aún puede usar el**flujo de URL manual**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute abre la URL de autorización de Google. -2. Después de autorizar, Google intenta redirigir a `localhost` (que falla en el servidor remoto) -3.**Copia la URL completa**de la barra de direcciones de tu navegador (incluso si la página no se carga) -4. Pegue esa URL en el campo que se muestra en el modo de conexión de OmniRoute. -5. Haga clic en**"Conectar"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Esto funciona porque el código de autorización en la URL es válido independientemente de si se cargó la página de redireccionamiento.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. - -🇧🇷 Versión en portugués#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Los proveedores**Antigravity**y**Gemini CLI**usan**Google OAuth 2.0**para autenticar. O Google exige que un `redirect_uri` usado sin flujo OAuth seja**exatamente**uma das URI pre-cadastradas en Google Cloud Console de la aplicación. +
+🇧🇷 Versão em Português -Como credenciales OAuth embutidas no OmniRoute están catastradas**apenas para `localhost`**. Cuando accede a OmniRoute en un servidor remoto (por ejemplo: `https://omniroute.meuservidor.com`), o Google envía una autenticación con:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Debe crear un**ID de cliente OAuth 2.0**en Google Cloud Console con un URI en su servidor.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Acceso a Google Cloud Console** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -**2. Llame a un nuevo ID de cliente OAuth 2.0** +**2. Crie um novo OAuth 2.0 Client ID** -- Haga clic en**"+ Crear credenciales"**→**"ID de cliente OAuth"** -- Tipo de aplicación:**"Aplicación web"** -- Nombre: escolha qualquer nome (por ejemplo: `OmniRoute Remote`) +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Agregar como URI de redireccionamiento autorizado** +**3. Adicione as Authorized Redirect URIs** -No hay campo**"URI de redireccionamiento autorizado"**, además:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Sustituye `seu-servidor.com` por el dominio o IP de tu servidor (incluye una porta si es necesaria, por ejemplo: `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. Salve y copie como credencial** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Después de abrir, Google mostrará**ID de cliente**y**Secreto de cliente**. +**5. Configure as variáveis de ambiente** -**5. Configurar como variáveis de ambiente** +No seu `.env` (ou nas variáveis de ambiente do Docker): -No seu `.env` (o las variaciones de ambiente de Docker):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1867,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reiniciar OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute - -```` +``` **7. Tente conectar novamente** -Panel → Proveedores → Antigravity (o Gemini CLI) → OAuth +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Agora o Google redirigirá correctamente para `https://seu-servidor.com/callback` y autenticação funcionará.--- +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. + +--- #### Workaround temporário (sem configurar credenciais próprias) -Si no quieres crear credenciales propias ahora, aún puedes usar el flujo**manual de URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. El OmniRoute abrirá una URL de autorización de Google -2. Después de autorizar, Google intentará redirigir a `localhost` (que no tiene servidor remoto) -3.**Copia una URL completa**de la barra de envío de tu navegador (también que a página no carregue) -4. Cole esa URL en el campo que aparece en el modo de conexión de OmniRoute -5. Haz clic en**"Conectar"** +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) +4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute +5. Clique em **"Connect"** -> Esta solución funciona porque el código de autorización de la URL es válido independiente de la redirección ter cargada o no.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1905,64 +2171,73 @@ Si no quieres crear credenciales propias ahora, aún puedes usar el flujo**manua ## 🛠️ Tech Stack - -Haga clic para ampliar los detalles de la pila tecnológica +
+Click to expand tech stack details --**Tiempo de ejecución**: Node.js 18–22 LTS (⚠️ Node.js 24+**no es compatible**; los archivos binarios nativos `better-sqlite3` son incompatibles) --**Idioma**: TypeScript 5.9 —**100% TypeScript**en `src/` y `open-sse/` (cero `any` en los módulos principales desde v2.0) --**Marco**: Next.js 16 + React 19 + Tailwind CSS 4 --**Base de datos**: LowDB (JSON) + SQLite (estado de dominio + registros de proxy + auditoría de MCP + decisiones de enrutamiento) --**Esquemas**: Zod (validación de E/S de herramienta MCP, contratos API) --**Protocolos**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Transmisión**: Eventos enviados por el servidor (SSE) --**Auth**: OAuth 2.0 (PKCE) + JWT + Claves API + Autorización con alcance MCP --**Pruebas**: Ejecutor de pruebas de Node.js + Vitest (más de 900 pruebas que incluyen unidad, integración, E2E) --**CI/CD**: Acciones de GitHub (publicación automática de npm + Docker Hub en el lanzamiento) --**Sitio web**: [omniroute.online](https://omniroute.online) --**Paquete**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Resiliencia**: disyuntor, retroceso exponencial, rebaño anti-truenos, suplantación de TLS, autocuración combinada automática
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Documentación -| Documento | Descripción | +| Document | Description | | ---------------------------------------------- | --------------------------------------------------- | -| [Guía del usuario](docs/USER_GUIDE.md) | Proveedores, combos, integración CLI, implementación | -| [Referencia de API](docs/API_REFERENCE.md) | Todos los puntos finales con ejemplos | -| [Servidor MCP](open-sse/mcp-server/README.md) | 16 herramientas MCP, configuraciones IDE, clientes Python/TS/Go | -| [Servidor A2A](src/lib/a2a/README.md) | Protocolo JSON-RPC 2.0, habilidades, streaming, gestión de tareas | -| [Motor de combinación automática](docs/auto-combo.md) | Puntuación de 6 factores, paquetes de modos, autocuración | -| [Solución de problemas](docs/TROUBLESHOOTING.md) | Problemas comunes y soluciones | -| [Arquitectura](docs/ARCHITECTURE.md) | Arquitectura del sistema e partes internas | -| [Contribuyendo](CONTRIBUYENDO.md) | Configuración y pautas de desarrollo | -| [Especificación de OpenAPI](docs/openapi.yaml) | Especificación OpenAPI 3.0 | -| [Política de seguridad](SECURITY.md) | Informes de vulnerabilidad y prácticas de seguridad | -| [Implementación de VM](docs/VM_DEPLOYMENT_GUIDE.md) | Guía completa: configuración de VM + nginx + Cloudflare | -| [Galería de funciones](docs/FEATURES.md) | Recorrido visual por el panel con capturas de pantalla | -| [Lista de verificación de lanzamiento](docs/RELEASE_CHECKLIST.md) | Pasos de validación previa al lanzamiento |--- +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute tiene**más de 210 funciones planificadas**en múltiples fases de desarrollo. Estas son las áreas clave: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Categoría | Funciones planificadas | Aspectos destacados | -| ----------------------- | ---------------- | -------------------------------------------------------------------------------------- | -| 🧠**Enrutamiento e inteligencia**| 25+ | Enrutamiento de latencia más baja, enrutamiento basado en etiquetas, verificación previa de cuotas, selección de cuentas P2C | -| 🔒**Seguridad y cumplimiento**| 20+ | Refuerzo SSRF, encubrimiento de credenciales, límite de velocidad por punto final, alcance de claves de administración | -| 📊**Observabilidad**| 15+ | Integración de OpenTelemetry, monitoreo de cuotas en tiempo real, seguimiento de costos por modelo | -| 🔄**Integraciones de proveedores**| 20+ | Registro de modelo dinámico, tiempos de reutilización de proveedores, Codex multicuenta, análisis de cuotas de Copilot | -| ⚡**Rendimiento**| 15+ | Capa de caché dual, caché de avisos, caché de respuestas, transmisión keepalive, API por lotes | -| 🌐**Ecosistema**| 10+ | API WebSocket, recarga en caliente de configuración, almacén de configuración distribuido, modo comercial |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**Integración OpenCode**: soporte de proveedor nativo para el IDE de codificación OpenCode AI -- 🔗**Integración TRAE**: soporte total para el marco de desarrollo de IA de TRAE -- 📦**API por lotes**: procesamiento por lotes asíncrono para solicitudes masivas -- 🎯**Enrutamiento basado en etiquetas**: enruta solicitudes basadas en etiquetas y metadatos personalizados -- 💰**Estrategia de menor costo**: seleccione automáticamente el proveedor más barato disponible +### 🔜 Coming Soon -> 📝 Especificaciones completas de funciones disponibles en [`docs/new-features/`](docs/new-features/) (217 especificaciones detalladas)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1970,18 +2245,20 @@ OmniRoute tiene**más de 210 funciones planificadas**en múltiples fases de desa ### How to Contribute -1. Bifurcar el repositorio -2. Crea tu rama de funciones (`git checkout -b feature/amazing-feature`) -3. Confirme sus cambios (`git commit -m 'Agregar característica sorprendente'`) -4. Empuje a la rama (`git push origin feature/amazing-feature`) -5. Abra una solicitud de extracción +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Consulte [CONTRIBUTING.md](CONTRIBUTING.md) para obtener pautas detalladas.### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1993,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Un agradecimiento especial a**[9router](https://github.com/decolua/9router)**de**[decolua](https://github.com/decolua)**, el proyecto original que inspiró esta bifurcación. OmniRoute se basa en esa increíble base con funciones adicionales, API multimodales y una reescritura completa de TypeScript. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Un agradecimiento especial a**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**: la implementación original de Go que inspiró este puerto de JavaScript.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Licencia -Licencia MIT: consulte [LICENCIA](LICENCIA) para obtener más detalles.--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/es/docs/ARCHITECTURE.md b/docs/i18n/es/docs/ARCHITECTURE.md index 5b58707141..1783363207 100644 --- a/docs/i18n/es/docs/ARCHITECTURE.md +++ b/docs/i18n/es/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Última actualización: 2026-03-28_## Executive Summary -OmniRoute es un panel y una puerta de enlace de enrutamiento de IA local creado en Next.js. -Proporciona un único punto final compatible con OpenAI (`/v1/*`) y enruta el tráfico a través de múltiples proveedores ascendentes con traducción, respaldo, actualización de tokens y seguimiento de uso. -Capacidades principales: +_Last updated: 2026-03-28_ -- Superficie API compatible con OpenAI para CLI/herramientas (28 proveedores) -- Traducción de solicitudes/respuestas entre formatos de proveedores. -- Modelo combinado de respaldo (secuencia multimodelo) -- Respaldo a nivel de cuenta (varias cuentas por proveedor) -- Gestión de conexión de proveedor de claves OAuth + API -- Generación de incrustaciones mediante `/v1/embeddings` (6 proveedores, 9 modelos) -- Generación de imágenes a través de `/v1/images/generaciones` (4 proveedores, 9 modelos) -- Piense en el análisis de etiquetas (`...`) para modelos de razonamiento -- Saneamiento de respuesta para una estricta compatibilidad con OpenAI SDK -- Normalización de roles (desarrollador → sistema, sistema → usuario) para compatibilidad entre proveedores -- Conversión de salida estructurada (json_schema → Gemini ResponseSchema) -- Persistencia local para proveedores, claves, alias, combos, configuraciones, precios. -- Seguimiento de uso/costos y registro de solicitudes -- Sincronización en la nube opcional para sincronización multidispositivo/estado -- Lista de IP permitidas/lista de bloqueo para control de acceso a API -- Pensando en la gestión del presupuesto (transferencia/automática/personalizada/adaptativa) -- Inyección rápida del sistema global -- Seguimiento de sesiones y toma de huellas digitales -- Limitación de tarifas mejorada por cuenta con perfiles específicos del proveedor -- Patrón de disyuntor para la resiliencia del proveedor -- Protección de rebaño anti-truenos con bloqueo mutex -- Caché de deduplicación de solicitudes basado en firmas -- Capa de dominio: disponibilidad del modelo, reglas de costos, política de respaldo, política de bloqueo -- Persistencia del estado del dominio (caché de escritura SQLite para respaldos, presupuestos, bloqueos, disyuntores) -- Motor de políticas para la evaluación centralizada de solicitudes (bloqueo → presupuesto → respaldo) -- Solicitar telemetría con agregación de latencia p50/p95/p99 -- ID de correlación (X-Request-Id) para seguimiento de un extremo a otro -- Registro de auditoría de cumplimiento con opción de exclusión por clave API -- Marco de evaluación para el aseguramiento de la calidad del LLM. -- Panel de interfaz de usuario de resiliencia con estado del disyuntor en tiempo real -- Proveedores modulares de OAuth (12 módulos individuales en `src/lib/oauth/providers/`) +## Executive Summary -Modelo de tiempo de ejecución principal: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- Las rutas de la aplicación Next.js en `src/app/api/*` implementan tanto las API del panel como las API de compatibilidad. -- Un núcleo de enrutamiento/SSE compartido en `src/sse/*` + `open-sse/*` maneja la ejecución, traducción, transmisión, respaldo y uso del proveedor.## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Tiempo de ejecución de la puerta de enlace local -- API de gestión de paneles -- Autenticación de proveedor y actualización de token -- Solicitar traducción y transmisión SSE -- Estado local + persistencia de uso. -- Orquestación de sincronización en la nube opcional### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Implementación del servicio en la nube detrás de `NEXT_PUBLIC_CLOUD_URL` -- Proveedor SLA/plano de control fuera del proceso local -- Los propios binarios CLI externos (Claude CLI, Codex CLI, etc.)## Dashboard Surface (Current) +### Out of Scope -Páginas principales en `src/app/(dashboard)/dashboard/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — inicio rápido + descripción general del proveedor -- `/dashboard/endpoint` — proxy de punto final + MCP + A2A + pestañas de punto final API -- `/dashboard/providers` — conexiones y credenciales de proveedores -- `/dashboard/combos` — estrategias combinadas, plantillas, reglas de enrutamiento de modelos -- `/dashboard/costs` — agregación de costos y visibilidad de precios -- `/dashboard/analytics` — análisis y evaluaciones de uso -- `/dashboard/limits` — controles de cuota/tasa -- `/dashboard/cli-tools`: incorporación de CLI, detección de tiempo de ejecución, generación de configuración -- `/dashboard/agents` — agentes ACP detectados + registro de agente personalizado -- `/dashboard/media` — área de juegos de imágenes/videos/música -- `/dashboard/search-tools` — historial y pruebas del proveedor de búsqueda -- `/dashboard/health`: tiempo de actividad, disyuntores, límites de velocidad -- `/dashboard/logs` — registros de solicitud/proxy/auditoría/consola -- `/dashboard/settings`: pestañas de configuración del sistema (general, enrutamiento, valores predeterminados combinados, etc.) -- `/dashboard/api-manager` — Ciclo de vida de la clave API y permisos del modelo## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -181,89 +194,99 @@ Management domains: ## 2) SSE + Translation Core -Módulos de flujo principales: +Main flow modules: -- Entrada: `src/sse/handlers/chat.ts` -- Orquestación central: `open-sse/handlers/chatCore.ts` -- Adaptadores de ejecución del proveedor: `open-sse/executors/*` -- Detección de formato/configuración del proveedor: `open-sse/services/provider.ts` -- Análisis/resolución del modelo: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Lógica alternativa de cuenta: `open-sse/services/accountFallback.ts` -- Registro de traducción: `open-sse/translator/index.ts` -- Transformaciones de flujo: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- Extracción/normalización de uso: `open-sse/utils/usageTracking.ts` -- Analizador de etiquetas Think: `open-sse/utils/thinkTagParser.ts` -- Controlador de incrustación: `open-sse/handlers/embeddings.ts` -- Registro de proveedores de incrustación: `open-sse/config/embeddingRegistry.ts` -- Manejador de generación de imágenes: `open-sse/handlers/imageGeneration.ts` -- Registro del proveedor de imágenes: `open-sse/config/imageRegistry.ts` -- Sanitización de respuestas: `open-sse/handlers/responseSanitizer.ts` -- Normalización de roles: `open-sse/services/roleNormalizer.ts` +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -Servicios (lógica de negocios): +Services (business logic): -- Selección/puntuación de cuenta: `open-sse/services/accountSelector.ts` -- Gestión del ciclo de vida del contexto: `open-sse/services/contextManager.ts` -- Aplicación del filtro IP: `open-sse/services/ipFilter.ts` -- Seguimiento de sesión: `open-sse/services/sessionManager.ts` -- Solicitar deduplicación: `open-sse/services/signatureCache.ts` -- Inyección de aviso del sistema: `open-sse/services/systemPrompt.ts` -- Pensando en la gestión del presupuesto: `open-sse/services/thinkingBudget.ts` -- Enrutamiento del modelo comodín: `open-sse/services/wildcardRouter.ts` -- Gestión de límites de tarifas: `open-sse/services/rateLimitManager.ts` -- Disyuntor: `open-sse/services/circuitBreaker.ts` +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -Módulos de capa de dominio: +Domain layer modules: -- Disponibilidad del modelo: `src/lib/domain/modelAvailability.ts` -- Reglas de costos/presupuestos: `src/lib/domain/costRules.ts` -- Política alternativa: `src/lib/domain/fallbackPolicy.ts` -- Resolución combinada: `src/lib/domain/comboResolver.ts` -- Política de bloqueo: `src/lib/domain/lockoutPolicy.ts` -- Motor de políticas: `src/domain/policyEngine.ts` — bloqueo centralizado → presupuesto → evaluación alternativa -- Catálogo de códigos de error: `src/lib/domain/errorCodes.ts` -- ID de solicitud: `src/lib/domain/requestId.ts` -- Recuperar tiempo de espera: `src/lib/domain/fetchTimeout.ts` -- Solicitar telemetría: `src/lib/domain/requestTelemetry.ts` -- Cumplimiento/auditoría: `src/lib/domain/compliance/index.ts` -- Corredor de evaluación: `src/lib/domain/evalRunner.ts` -- Persistencia del estado del dominio: `src/lib/db/domainState.ts` — SQLite CRUD para cadenas de respaldo, presupuestos, historial de costos, estado de bloqueo, disyuntores +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -Módulos de proveedor de OAuth (12 archivos individuales en `src/lib/oauth/providers/`): +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -- Índice de registro: `src/lib/oauth/providers/index.ts` -- Proveedores individuales: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` -- Contenedor delgado: `src/lib/oauth/providers.ts` — reexportaciones desde módulos individuales## 3) Persistence Layer +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -Base de datos de estado primario (SQLite): +## 3) Persistence Layer -- Infraestructura principal: `src/lib/db/core.ts` (better-sqlite3, migraciones, WAL) -- Reexportación de fachada: `src/lib/localDb.ts` (capa delgada de compatibilidad para quienes llaman) -- archivo: `${DATA_DIR}/storage.sqlite` (o `$XDG_CONFIG_HOME/omniroute/storage.sqlite` cuando está configurado, en caso contrario `~/.omniroute/storage.sqlite`) -- entidades (tablas + espacios de nombres KV): conexiones de proveedor, nodos de proveedor, alias de modelo, combos, claves de API, configuración, precios,**modelos personalizados**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +Primary state DB (SQLite): -Persistencia de uso: +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -- fachada: `src/lib/usageDb.ts` (módulos descompuestos en `src/lib/usage/*`) -- Tablas SQLite en `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` -- Los artefactos de archivos opcionales permanecen para compatibilidad/depuración (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- Los archivos JSON heredados se migran a SQLite mediante migraciones de inicio cuando están presentes +Usage persistence: -Base de datos de estado de dominio (SQLite): +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- `src/lib/db/domainState.ts` — Operaciones CRUD para el estado del dominio -- Tablas (creadas en `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- Patrón de caché de escritura simultánea: los mapas en memoria tienen autoridad en tiempo de ejecución; las mutaciones se escriben sincrónicamente en SQLite; El estado se restaura desde la base de datos en el arranque en frío.## 4) Auth + Security Surfaces +Domain State DB (SQLite): -- Autenticación de cookies del panel: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- Generación/verificación de clave API: `src/shared/utils/apiKey.ts` -- Los secretos del proveedor persistieron en las entradas de `providerConnections` -- Soporte de proxy saliente a través de `open-sse/utils/proxyFetch.ts` (env vars) y `open-sse/utils/networkProxy.ts` (configurable por proveedor o global)## 5) Cloud Sync +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start -- Inicio del programador: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Tarea periódica: `src/shared/services/cloudSyncScheduler.ts` -- Tarea periódica: `src/shared/services/modelSyncScheduler.ts` -- Ruta de control: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -340,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Las decisiones de respaldo están impulsadas por `open-sse/services/accountFallback.ts` utilizando códigos de estado y heurísticas de mensajes de error. El enrutamiento combinado agrega una protección adicional: los 400 con alcance del proveedor, como los errores de validación de roles y bloques de contenido ascendentes, se tratan como errores del modelo local, por lo que los destinos combinados posteriores aún pueden ejecutarse.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -370,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -La actualización durante el tráfico en vivo se ejecuta dentro de `open-sse/handlers/chatCore.ts` a través del ejecutor `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -402,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -La sincronización periódica la activa "CloudSyncScheduler" cuando la nube está habilitada.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -503,12 +532,14 @@ erDiagram } ``` -Archivos de almacenamiento físico: +Physical storage files: -- Base de datos de ejecución principal: `${DATA_DIR}/storage.sqlite` -- solicitar líneas de registro: `${DATA_DIR}/log.txt` (artefacto de compatibilidad/depuración) -- archivos de carga útil de llamadas estructuradas: `${DATA_DIR}/call_logs/` -- traductor opcional/solicitar sesiones de depuración: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -543,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: API de compatibilidad -- `src/app/api/v1/providers/[provider]/*`: rutas dedicadas por proveedor (chat, incrustaciones, imágenes) -- `src/app/api/providers*`: proveedor CRUD, validación, pruebas -- `src/app/api/provider-nodes*`: gestión personalizada de nodos compatibles -- `src/app/api/provider-models`: gestión de modelos personalizados (CRUD) -- `src/app/api/models/route.ts`: API de catálogo de modelos (alias + modelos personalizados) -- `src/app/api/oauth/*`: OAuth/flujos de código de dispositivo -- `src/app/api/keys*`: ciclo de vida de la clave API local -- `src/app/api/models/alias`: gestión de alias -- `src/app/api/combos*`: gestión de combos alternativos -- `src/app/api/pricing`: anulación de precios para el cálculo de costos -- `src/app/api/settings/proxy`: configuración del proxy (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: prueba de conectividad de proxy saliente (POST) -- `src/app/api/usage/*`: API de uso y registros -- `src/app/api/sync/*` + `src/app/api/cloud/*`: sincronización en la nube y ayudantes orientados a la nube -- `src/app/api/cli-tools/*`: escritores/comprobadores de configuración CLI local -- `src/app/api/settings/ip-filter`: lista de IP permitidas/lista de bloqueo (GET/PUT) -- `src/app/api/settings/thinking-budget`: configuración del presupuesto del token de pensamiento (GET/PUT) -- `src/app/api/settings/system-prompt`: indicador global del sistema (GET/PUT) -- `src/app/api/sessions`: listado de sesiones activas (GET) -- `src/app/api/rate-limits`: estado del límite de tasa por cuenta (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: análisis de solicitudes, manejo combinado, bucle de selección de cuentas -- `open-sse/handlers/chatCore.ts`: traducción, envío del ejecutor, reintento/actualización, configuración de flujo -- `open-sse/executors/*`: comportamiento de formato y red específico del proveedor### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: registro y orquestación de traductores -- Solicitar traductores: `open-sse/translator/request/*` -- Traductores de respuesta: `open-sse/translator/response/*` -- Constantes de formato: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: configuración/estado persistente y persistencia de dominio en SQLite -- `src/lib/localDb.ts`: reexportación de compatibilidad para módulos DB -- `src/lib/usageDb.ts`: fachada de historial de uso/registros de llamadas encima de las tablas SQLite## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Cada proveedor tiene un ejecutor especializado que extiende `BaseExecutor` (en `open-sse/executors/base.ts`), que proporciona creación de URL, construcción de encabezados, reintentos con retroceso exponencial, enlaces de actualización de credenciales y el método de orquestación `execute()`. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Ejecutor | Proveedor(es) | Manejo Especial | -| ------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | --------------------------------------------------------------------------------------------- | -| `Ejecutor predeterminado` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Configuración dinámica de URL/encabezado por proveedor | -| `AntigravityExecutor` | Antigravedad de Google | ID personalizados de proyecto/sesión, reintento después del análisis | -| `CodexExecutor` | Códice OpenAI | Inyecta instrucciones del sistema, fuerza el esfuerzo de razonamiento | -| `CursorEjecutor` | Cursor IDE | Protocolo ConnectRPC, codificación Protobuf, solicitud de firma mediante suma de comprobación | -| `GithubExecutor` | Copiloto de GitHub | Actualización del token Copilot, encabezados que imitan VSCode | -| `KiroExecutor` | AWS CodeWhisperer/Kiro | Formato binario de AWS EventStream → Conversión SSE | -| `GeminiCLIExecutor` | Géminis CLI | Ciclo de actualización del token OAuth de Google | +### Persistence -Todos los demás proveedores (incluidos los nodos compatibles personalizados) utilizan `DefaultExecutor`.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Proveedor | Formato | Autenticación | Corriente | Sin transmisión | Actualización de token | API de uso | -| ---------------------- | ----------------- | ---------------------------------- | --------------------------- | --------------- | ---------------------- | ------------------------- | ------------------------------ | -| Claudio | claudio | Clave API/OAuth | ✅ | ✅ | ✅ | ⚠️ Solo administrador | -| Géminis | géminis | Clave API/OAuth | ✅ | ✅ | ✅ | ⚠️ Consola en la nube | -| Géminis CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Consola en la nube | -| Antigravedad | antigravedad | OAuth | ✅ | ✅ | ✅ | ✅ API de cuota completa | -| Abierta AI | abierto | Clave API | ✅ | ✅ | ❌ | ❌ | -| Códice | respuestas-openai | OAuth | ✅ forzado | ❌ | ✅ | ✅ Límites de tarifas | -| Copiloto de GitHub | abierto | OAuth + Token de copiloto | ✅ | ✅ | ✅ | ✅ Instantáneas de cuotas | -| Cursores | cursor | Suma de comprobación personalizada | ✅ | ✅ | ❌ | ❌ | -| kiro | kiro | AWS SSO OIDC | ✅ (Transmisión de eventos) | ❌ | ✅ | ✅ Límites de uso | -| Qwen | abierto | OAuth | ✅ | ✅ | ✅ | ⚠️ Por solicitud | -| Qoder | abierto | OAuth (básico) | ✅ | ✅ | ✅ | ⚠️ Por solicitud | -| Enrutador abierto | abierto | Clave API | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | claudio | Clave API | ✅ | ✅ | ❌ | ❌ | -| Búsqueda profunda | abierto | Clave API | ✅ | ✅ | ❌ | ❌ | -| Groq | abierto | Clave API | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | abierto | Clave API | ✅ | ✅ | ❌ | ❌ | -| Mistral | abierto | Clave API | ✅ | ✅ | ❌ | ❌ | -| Perplejidad | abierto | Clave API | ✅ | ✅ | ❌ | ❌ | -| Juntos IA | abierto | Clave API | ✅ | ✅ | ❌ | ❌ | -| Fuegos artificiales AI | abierto | Clave API | ✅ | ✅ | ❌ | ❌ | -| Cerebras | abierto | Clave API | ✅ | ✅ | ❌ | ❌ | -| Coherir | abierto | Clave API | ✅ | ✅ | ❌ | ❌ | -| NIM de NVIDIA | abierto | Clave API | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Los formatos de origen detectados incluyen: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. + +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | + +All other providers (including custom compatible nodes) use the `DefaultExecutor`. + +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: - `openai` -- `respuestas openai` -- `claudio` -- `géminis` +- `openai-responses` +- `claude` +- `gemini` -Los formatos de destino incluyen: +Target formats include: -- Chat/Respuestas de OpenAI -- Claudio -- Géminis/Gemini-CLI/sobre antigravedad - -Kiro -- Cursores +- OpenAI chat/Responses +- Claude +- Gemini/Gemini-CLI/Antigravity envelope +- Kiro +- Cursor -Las traducciones utilizan**OpenAI como formato central**; todas las conversiones pasan por OpenAI como formato intermedio:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Las traducciones se seleccionan dinámicamente según la forma de la carga útil de origen y el formato de destino del proveedor. +Additional processing layers in the translation pipeline: -Capas de procesamiento adicionales en el proceso de traducción: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Desinfección de respuestas**: elimina los campos no estándar de las respuestas en formato OpenAI (tanto en streaming como sin streaming) para garantizar el estricto cumplimiento del SDK. --**Normalización de roles**: convierte `desarrollador` → `sistema` para objetivos que no son OpenAI; fusiona `sistema` → `usuario` para modelos que rechazan el rol del sistema (GLM, ERNIE) --**Extracción de etiquetas Think**: analiza los bloques `...` del contenido en el campo `reasoning_content` --**Salida estructurada**: convierte OpenAI `response_format.json_schema` en `responseMimeType` + `responseSchema` de Gemini.## Supported API Endpoints +## Supported API Endpoints -| Punto final | Formato | Manejador | +| Endpoint | Format | Handler | | -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | -| `POST /v1/chat/compleciones` | Chat abierto de IA | `src/sse/handlers/chat.ts` | -| `POST /v1/mensajes` | Mensajes de Claude | Mismo controlador (detectado automáticamente) | -| `POST /v1/respuestas` | Respuestas de OpenAI | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/incrustaciones` | Incrustaciones de OpenAI | `open-sse/handlers/embeddings.ts` | -| `GET /v1/incrustaciones` | Listado de modelos | Ruta API | -| `POST /v1/imagenes/generaciones` | Imágenes de OpenAI | `open-sse/handlers/imageGeneration.ts` | -| `GET /v1/images/generaciones` | Listado de modelos | Ruta API | -| `POST /v1/proveedores/{proveedor}/chat/completions` | Chat abierto de IA | Dedicado por proveedor con validación de modelo | -| `POST /v1/proveedores/{proveedor}/incrustaciones` | Incrustaciones de OpenAI | Dedicado por proveedor con validación de modelo | -| `POST /v1/proveedores/{proveedor}/images/generaciones` | Imágenes de OpenAI | Dedicado por proveedor con validación de modelo | -| `POST /v1/mensajes/count_tokens` | Recuento de fichas de Claude | Ruta API | -| `OBTENER /v1/modelos` | Lista de modelos OpenAI | Ruta API (chat + incrustación + imagen + modelos personalizados) | -| `OBTENER /api/modelos/catalog` | Catálogo | Todos los modelos agrupados por proveedor + tipo | -| `POST /v1beta/models/*:streamGenerateContent` | Nativo de Géminis | Ruta API | -| `OBTENER/PONER/BORRAR /api/settings/proxy` | Configuración de proxy | Configuración del proxy de red | -| `POST /api/configuración/proxy/prueba` | Conectividad de proxy | Punto final de prueba de conectividad/estado del proxy | -| `GET/POST/DELETE /api/provider-models` | Modelos de proveedores | Metadatos del modelo de proveedor que respaldan los modelos disponibles personalizados y administrados |## Bypass Handler +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -El controlador de omisión (`open-sse/utils/bypassHandler.ts`) intercepta solicitudes "desechables" conocidas de Claude CLI (pings de preparación, extracciones de títulos y recuentos de tokens) y devuelve una**respuesta falsa**sin consumir tokens del proveedor ascendente. Esto se activa solo cuando "User-Agent" contiene "claude-cli".## Request Logger Pipeline +## Bypass Handler -El registrador de solicitudes (`open-sse/utils/requestLogger.ts`) proporciona un canal de registro de depuración de 7 etapas, deshabilitado de forma predeterminada, habilitado a través de `ENABLE_REQUEST_LOGS=true`:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Los archivos se escriben en `/logs//` para cada sesión de solicitud.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- tiempo de reutilización de la cuenta del proveedor en errores transitorios/de tasa/autenticación -- respaldo de la cuenta antes de fallar la solicitud -- retroceso del modelo combinado cuando se agota la ruta del modelo/proveedor actual## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- verificación previa y actualización con reintento para proveedores actualizables -- Reintento 401/403 después de un intento de actualización en la ruta principal## 3) Stream Safety +## 2) Token Expiry -- controlador de flujo con reconocimiento de desconexión -- flujo de traducción con descarga de final de flujo y manejo `[DONE]` -- reserva de estimación de uso cuando faltan metadatos de uso del proveedor## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- Aparecen errores de sincronización pero el tiempo de ejecución local continúa -- El programador tiene una lógica con capacidad de reintento, pero la ejecución periódica actualmente llama a la sincronización de un solo intento de forma predeterminada.## 5) Data Integrity +## 3) Stream Safety -- Migraciones de esquema SQLite y enlaces de actualización automática al inicio -- JSON heredado → ruta de compatibilidad de migración SQLite## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Fuentes de visibilidad en tiempo de ejecución: +## 4) Cloud Sync Degradation -- registros de consola desde `src/sse/utils/logger.ts` -- agregados de uso por solicitud en SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- capturas de carga útil detalladas en cuatro etapas en SQLite (`request_detail_logs`) cuando `settings.detailed_logs_enabled=true` -- registro de estado de solicitud textual en `log.txt` (opcional/compatible) -- registros de traducción/solicitud profunda opcionales en `logs/` cuando `ENABLE_REQUEST_LOGS=true` -- puntos finales de uso del panel (`/api/usage/*`) para el consumo de UI +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -La captura de carga útil de solicitud detallada almacena hasta cuatro etapas de carga útil JSON por llamada enrutada: +## 5) Data Integrity -- solicitud sin procesar recibida del cliente -- solicitud traducida realmente enviada en sentido ascendente -- respuesta del proveedor reconstruida como JSON; las respuestas transmitidas se compactan en el resumen final más los metadatos de la transmisión -- respuesta final del cliente devuelta por OmniRoute; las respuestas transmitidas se almacenan en el mismo formulario de resumen compacto## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- JWT secret (`JWT_SECRET`) protege la verificación/firma de cookies de sesión del panel -- El arranque de contraseña inicial (`INITIAL_PASSWORD`) debe configurarse explícitamente para el aprovisionamiento de primera ejecución. -- La clave API HMAC secreta (`API_KEY_SECRET`) protege el formato de clave API local generado -- Los secretos del proveedor (claves/tokens de API) se conservan en la base de datos local y deben protegerse a nivel del sistema de archivos. -- Los puntos finales de sincronización en la nube se basan en la semántica de autenticación de clave API + ID de máquina## Environment and Runtime Matrix +## Observability and Operational Signals -Variables de entorno utilizadas activamente por el código: +Runtime visibility sources: -- Aplicación/autenticación: `JWT_SECRET`, `INITIAL_PASSWORD` -- Almacenamiento: `DATA_DIR` -- Comportamiento de nodo compatible: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Anulación de la base de almacenamiento opcional (Linux/macOS cuando `DATA_DIR` no está configurado): `XDG_CONFIG_HOME` -- Hashing de seguridad: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Registro: `ENABLE_REQUEST_LOGS` -- Sincronización/URL en la nube: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- Proxy saliente: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` y variantes en minúsculas -- Marcas de características de SOCKS5: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- Ayudantes de plataforma/tiempo de ejecución (no configuración específica de la aplicación): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` y `localDb` comparten la misma política de directorio base (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) con la migración de archivos heredados. -2. `/api/v1/route.ts` delega al mismo generador de catálogo unificado utilizado por `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) para evitar la deriva semántica. -3. El registrador de solicitudes escribe encabezados/cuerpo completo cuando está habilitado; trate el directorio de registro como confidencial. -4. El comportamiento de la nube depende de la `NEXT_PUBLIC_BASE_URL` correcta y de la accesibilidad del punto final de la nube. -5. El directorio `open-sse/` se publica como `@omniroute/open-sse`**paquete de espacio de trabajo npm**. El código fuente lo importa a través de `@omniroute/open-sse/...` (resuelto por Next.js `transpilePackages`). Las rutas de archivo en este documento todavía usan el nombre de directorio `open-sse/` para mantener la coherencia. -6. Los gráficos en el panel utilizan**Recharts**(basados ​​en SVG) para visualizaciones analíticas interactivas y accesibles (gráficos de barras de uso de modelos, tablas de desglose de proveedores con tasas de éxito). -7. Las pruebas E2E utilizan**Playwright**(`tests/e2e/`), se ejecutan mediante `npm run test:e2e`. Las pruebas unitarias utilizan**ejecutor de pruebas Node.js**(`tests/unit/`), se ejecutan a través de `npm run test:unit`. El código fuente bajo `src/` es**TypeScript**(`.ts`/`.tsx`); el espacio de trabajo `open-sse/` sigue siendo JavaScript (`.js`). -8. La página de configuración está organizada en 5 pestañas: Seguridad, Enrutamiento (6 estrategias globales: completar primero, por turnos, p2c, aleatorio, menos utilizado, de costo optimizado), Resiliencia (límites de velocidad editables, disyuntor, políticas), IA (presupuesto pensado, aviso del sistema, caché de avisos), Avanzado (proxy).## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- Compilación desde la fuente: `npm run build` -- Crear imagen de Docker: `docker build -t omniroute.` -- Iniciar el servicio y verificar: -- `OBTENER /api/configuración` -- `OBTENER /api/v1/modelos` -- La URL base de destino de CLI debe ser `http://:20128/v1` cuando `PORT=20128` +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` +- `GET /api/v1/models` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/es/docs/FEATURES.md b/docs/i18n/es/docs/FEATURES.md index 6a8945c517..ca348861f0 100644 --- a/docs/i18n/es/docs/FEATURES.md +++ b/docs/i18n/es/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Guía visual de cada sección del panel de OmniRoute.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Administre las conexiones de proveedores de IA: proveedores de OAuth (Claude Code, Codex, Gemini CLI), proveedores de claves API (Groq, DeepSeek, OpenRouter) y proveedores gratuitos (Qoder, Qwen, Kiro). Las cuentas Kiro incluyen seguimiento del saldo de crédito: créditos restantes, asignación total y fecha de renovación visibles en Panel → Uso.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Cree combinaciones de enrutamiento de modelos con 6 estrategias: prioridad, ponderada, por turnos, aleatoria, menos utilizada y de costo optimizado. Cada combo encadena múltiples modelos con respaldo automático e incluye plantillas rápidas y comprobaciones de preparación.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Análisis de uso integral con consumo de tokens, estimaciones de costos, mapas de actividad, gráficos de distribución semanal y desgloses por proveedor.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Monitoreo en tiempo real: tiempo de actividad, memoria, versión, percentiles de latencia (p50/p95/p99), estadísticas de caché y estados de los disyuntores del proveedor.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Cuatro modos para depurar traducciones de API:**Playground**(convertidor de formato),**Chat Tester**(solicitudes en vivo),**Test Bench**(pruebas por lotes) y**Live Monitor**(transmisión en tiempo real).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Pruebe cualquier modelo directamente desde el tablero. Seleccione proveedor, modelo y punto final, escriba mensajes con Monaco Editor, transmita respuestas en tiempo real, cancele la transmisión a mitad de camino y vea métricas de tiempo.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Temas de colores personalizables para todo el tablero. Elija entre 7 colores preestablecidos (coral, azul, rojo, verde, violeta, naranja, cian) o cree un tema personalizado eligiendo cualquier color hexadecimal. Admite modo claro, oscuro y de sistema.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Panel de configuración completo con pestañas: +Comprehensive settings panel with tabs: --**General**— Almacenamiento del sistema, gestión de copias de seguridad (exportación/importación de base de datos) -**Apariencia**: selector de tema (oscuro/claro/sistema), ajustes preestablecidos de tema de color y colores personalizados, visibilidad del registro de estado, controles de visibilidad de elementos de la barra lateral -**Seguridad**: protección API de endpoints, bloqueo de proveedores personalizado, filtrado de IP, información de sesión -**Enrutamiento**: alias de modelo, degradación de tareas en segundo plano -**Resiliencia**: persistencia del límite de velocidad, ajuste de disyuntores, desactivación automática de cuentas prohibidas, monitoreo de vencimiento del proveedor -**Avanzado**: anulaciones de configuración, seguimiento de auditoría de configuración, modo de degradación alternativa![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Configuración con un clic para herramientas de codificación de IA: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continuar, Cursor y Factory Droid. Incluye aplicación/restablecimiento de configuración automatizada, perfiles de conexión y mapeo de modelos.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Panel para descubrir y administrar agentes CLI. Muestra una cuadrícula de 14 agentes integrados (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) con: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Estado de instalación**: Instalado/No encontrado con detección de versión -**Insignias de protocolo**: stdio, HTTP, etc. -**Agentes personalizados**: registre cualquier herramienta CLI a través del formulario (nombre, binario, comando de versión, argumentos de generación) -**CLI Fingerprint Matching**: alternancia por proveedor para hacer coincidir las firmas de solicitud CLI nativas, lo que reduce el riesgo de prohibición y preserva la IP del proxy.--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Genere imágenes, videos y música desde el tablero. Admite OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open y MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Registro de solicitudes en tiempo real con filtrado por proveedor, modelo, cuenta y clave API. Muestra códigos de estado, uso de token, latencia y detalles de respuesta.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Su punto final API unificado con desglose de capacidades: finalización de chat, API de respuestas, incrustaciones, generación de imágenes, reclasificación, transcripción de audio, texto a voz, moderaciones y claves API registradas. Integración de Cloudflare Quick Tunnel y soporte de proxy en la nube para acceso remoto.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -Cree, alcance y revoque claves API. Cada clave se puede restringir a modelos/proveedores específicos con acceso completo o permisos de solo lectura. Gestión visual de claves con seguimiento de uso.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Seguimiento de acciones administrativas con filtrado por tipo de acción, actor, objetivo, dirección IP y marca de tiempo. Historial completo de eventos de seguridad.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Aplicación de escritorio Native Electron para Windows, macOS y Linux. Ejecute OmniRoute como una aplicación independiente con integración en la bandeja del sistema, soporte sin conexión, actualización automática e instalación con un solo clic. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Características clave: +Key features: -- Sondeo de preparación del servidor (no hay pantalla en blanco durante el arranque en frío) -- Bandeja del sistema con gestión de puertos. -- Política de seguridad de contenidos -- Cerradura de instancia única -- Actualización automática al reiniciar -- UI condicionada a la plataforma (semáforos de macOS, barra de título predeterminada de Windows/Linux) -- Paquete de compilación de Electron reforzado: los `node_modules' vinculados simbólicamente en el paquete independiente se detectan y rechazan antes del empaquetado, lo que evita la dependencia del tiempo de ejecución en la máquina de compilación (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Consulte [`electron/README.md`](../electron/README.md) para obtener la documentación completa. +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/es/docs/TROUBLESHOOTING.md b/docs/i18n/es/docs/TROUBLESHOOTING.md index 02011c3e25..4e7d35330d 100644 --- a/docs/i18n/es/docs/TROUBLESHOOTING.md +++ b/docs/i18n/es/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -Problemas comunes y soluciones para OmniRoute.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Problema | Solución | -| ------------------------------------------ | ---------------------------------------------------------------------------------------------- | --- | -| El primer inicio de sesión no funciona | Establezca `INITIAL_PASSWORD` en `.env` (sin valor predeterminado codificado) | -| El panel se abre en el puerto incorrecto | Establezca `PORT=20128` y `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| No hay registros de solicitudes en `logs/` | Establezca `ENABLE_REQUEST_LOGS = verdadero` | -| EACCES: permiso denegado | Establezca `DATA_DIR=/path/to/writable/dir` para anular `~/.omniroute` | -| La estrategia de enrutamiento no se guarda | Actualización a v1.4.11+ (corrección del esquema Zod para la persistencia de la configuración) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Causa:**Cuota de proveedor agotada. +**Cause:** Provider quota exhausted. -**Arreglo:** +**Fix:** -1. Verifique el rastreador de cuotas del panel -2. Utilice un combo con niveles alternativos -3. Cambiar al nivel más barato/gratuito### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Causa:**Cuota de suscripción agotada. +### Rate Limiting -**Arreglo:** +**Cause:** Subscription quota exhausted. -- Agregar respaldo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Utilice GLM/MiniMax como copia de seguridad económica### OAuth Token Expired +**Fix:** -OmniRoute actualiza automáticamente los tokens. Si los problemas persisten: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Panel de control → Proveedor → Reconectar -2. Eliminar y volver a agregar la conexión del proveedor.--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Verifique que `BASE_URL` apunte a su instancia en ejecución (por ejemplo, `http://localhost:20128`) -2. Verifique que `CLOUD_URL` apunte a su punto final en la nube (por ejemplo, `https://omniroute.dev`). -3. Mantenga los valores `NEXT_PUBLIC_*` alineados con los valores del lado del servidor### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Síntoma:**`Token inesperado 'd'...` en el punto final de la nube para llamadas que no son de transmisión. +### Cloud `stream=false` Returns 500 -**Causa:**Upstream devuelve la carga útil SSE mientras que el cliente espera JSON. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Solución alternativa:**Utilice `stream=true` para llamadas directas en la nube. El tiempo de ejecución local incluye el respaldo SSE → JSON.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Cree una clave nueva desde el panel local (`/api/keys`) -2. Ejecute la sincronización en la nube: Habilitar nube → Sincronizar ahora -3. Las claves antiguas/no sincronizadas aún pueden devolver "401" en la nube--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Verifique los campos de tiempo de ejecución: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. Para el modo portátil: use el destino de imagen `runner-cli` (CLI incluidas) -3. Para el modo de montaje del host: configure `CLI_EXTRA_PATHS` y monte el directorio bin del host como de solo lectura -4. Si `installed=true` y `runnable=false`: se encontró el binario pero falló la verificación de estado### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Verifique las estadísticas de uso en Panel → Uso -2. Cambie el modelo principal a GLM/MiniMax -3. Utilice el nivel gratuito (Gemini CLI, Qoder) para tareas no críticas -4. Establezca presupuestos de costos por clave API: Panel → Claves API → Presupuesto--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Establezca `ENABLE_REQUEST_LOGS=true` en su archivo `.env`. Los registros aparecen en el directorio `logs/`.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,67 +178,95 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Estado principal: `${DATA_DIR}/storage.sqlite` (proveedores, combos, alias, claves, configuraciones) -- Uso: tablas SQLite en `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + opcional `${DATA_DIR}/log.txt` y `${DATA_DIR}/call_logs/` -- Solicitar registros: `/logs/...` (cuando `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Cuando el disyuntor de un proveedor está ABIERTO, las solicitudes se bloquean hasta que expire el tiempo de reutilización. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Arreglo:** +**Fix:** -1. Vaya a**Panel → Configuración → Resiliencia** -2. Verifique la tarjeta del disyuntor del proveedor afectado. -3. Haga clic en**Restablecer todo**para borrar todos los interruptores o espere a que expire el tiempo de reutilización. -4. Verifique que el proveedor esté realmente disponible antes de restablecer### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Si un proveedor ingresa repetidamente al estado ABIERTO: +### Provider keeps tripping the circuit breaker -1. Marque**Panel → Estado → Estado del proveedor**para ver el patrón de error. -2. Vaya a**Configuración → Resiliencia → Perfiles de proveedores**y aumente el umbral de falla. -3. Verifique si el proveedor ha cambiado los límites de API o requiere una nueva autenticación. -4. Revise la telemetría de latencia: una latencia alta puede causar fallas basadas en el tiempo de espera--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Asegúrate de estar usando el prefijo correcto: `deepgram/nova-3` o `assemblyai/best` -- Verifique que el proveedor esté conectado en**Panel → Proveedores**### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Verifique los formatos de audio admitidos: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- Verifique que el tamaño del archivo esté dentro de los límites del proveedor (normalmente < 25 MB) -- Verifique la validez de la clave API del proveedor en la tarjeta del proveedor--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Utilice**Panel → Traductor**para depurar problemas de traducción de formato: +Use **Dashboard → Translator** to debug format translation issues: -| Modo | Cuándo utilizar | -| -------------------------- | ------------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Parque infantil** | Compare formatos de entrada/salida uno al lado del otro: pegue una solicitud fallida para ver cómo se traduce | -| **Probador de chat** | Envíe mensajes en vivo e inspeccione la carga útil completa de solicitud/respuesta, incluidos los encabezados | -| **Banco de pruebas** | Ejecute pruebas por lotes en combinaciones de formatos para encontrar qué traducciones no funcionan | -| **Monitorización en vivo** | Observe el flujo de solicitudes en tiempo real para detectar problemas de traducción intermitentes | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Las etiquetas de pensamiento no aparecen**: compruebe si el proveedor objetivo apoya el pensamiento y la configuración del presupuesto de pensamiento. -**Caídas de llamadas a herramientas**: algunas traducciones de formatos pueden eliminar campos no admitidos; verificar en modo Patio de Juegos -**Falta el mensaje del sistema**: Claude y Gemini manejan los mensajes del sistema de manera diferente; comprobar la salida de la traducción -**El SDK devuelve una cadena sin formato en lugar de un objeto**— Corregido en v1.1.0: el desinfectante de respuesta ahora elimina los campos no estándar (`x_groq`, `usage_breakdown`, etc.) que causan fallas de validación de Pydantic en el SDK de OpenAI -**GLM/ERNIE rechaza la función `sistema`**— Corregido en v1.1.0: el normalizador de funciones fusiona automáticamente mensajes del sistema con mensajes de usuario para modelos incompatibles -**Rol de "desarrollador" no reconocido**- Corregido en v1.1.0: convertido automáticamente a "sistema" para proveedores que no son OpenAI -**`json_schema` no funciona con Gemini**— Corregido en v1.1.0: `response_format` ahora se convierte a `responseMimeType` + `responseSchema` de Gemini--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- El límite de velocidad automático solo se aplica a los proveedores de claves API (no a OAuth/suscripción) -- Verifique que**Configuración → Resiliencia → Perfiles de proveedores**tenga habilitado el límite de tasa automática -- Verifique si el proveedor devuelve códigos de estado `429` o encabezados `Reintentar después`### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -Los perfiles de proveedor admiten estas configuraciones: +### Tuning exponential backoff --**Retraso base**: tiempo de espera inicial después del primer fallo (predeterminado: 1 s) -**Retraso máximo**: límite máximo de tiempo de espera (predeterminado: 30 segundos) -**Multiplicador**: cuánto aumentar el retraso por falla consecutiva (predeterminado: 2x)### Anti-thundering herd +Provider profiles support these settings: -Cuando muchas solicitudes simultáneas llegan a un proveedor de velocidad limitada, OmniRoute utiliza mutex + limitación de velocidad automática para serializar solicitudes y evitar fallas en cascada. Esto es automático para los proveedores de claves API.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) @@ -199,4 +305,8 @@ You can ignore this section if you do not run RAG or agent pipelines behind Omni ## Still Stuck? --**Problemas de GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Arquitectura**: consulte [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) para obtener detalles internos -**Referencia de API**: consulte [`docs/API_REFERENCE.md`](API_REFERENCE.md) para todos los puntos finales -**Panel de estado**: marque**Panel → Salud**para ver el estado del sistema en tiempo real -**Traductor**: use**Panel → Traductor**para depurar problemas de formato +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/es/llm.txt b/docs/i18n/es/llm.txt new file mode 100644 index 0000000000..d07f7bfa97 --- /dev/null +++ b/docs/i18n/es/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Español) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Resumen + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Seguridad +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/fi/README.md b/docs/i18n/fi/README.md index a2da873826..411a4e1a61 100644 --- a/docs/i18n/fi/README.md +++ b/docs/i18n/fi/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Universaali API-välityspalvelin – yksi päätepiste, yli 60 palveluntarjoajaa, nolla seisokkiaikaa. Nyt mukana**MCP-palvelin (25 työkalua)**,**A2A-protokolla**,**muisti/taitojärjestelmät**ja**electron-työpöytäsovellus**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Pikaviestien loppuun saattaminen • upotukset • kuvien luonti • video • musiikki • ääni • uudelleensijoitus •**verkkohaku**• MCP-palvelin • A2A-protokolla • 100 % TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Universaali API-välityspalvelin – yksi päätepiste, yli 60 palveluntarjoaja [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Verkkosivusto](https://omniroute.online) • [🚀 Pika-aloitus](#-pika-aloitus) • [💡 Ominaisuudet](#-avainominaisuudet) • [📖 Docs](#-dokumentaatio) • [💰 Hinnoittelu](#-hinnoittelu-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Saatavilla:**🇺🇸 [Englanti](README.md) | 🇧🇷 [Português (Brasilia)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Tanska](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugali)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,553 +60,629 @@ _Universaali API-välityspalvelin – yksi päätepiste, yli 60 palveluntarjoaja ## 📸 Dashboard Preview - -Napsauta nähdäksesi hallintapaneelin kuvakaappaukset +
+Click to see dashboard screenshots -| Sivu | Kuvakaappaus | -| ---------------- | -------------------------------------------------- | ---------- | -| **Tarjoajat** | ![Providers](docs/screenshots/01-providers.png) | -| **Yhdistelmät** | ![Combos](docs/screenshots/02-combos.png) | -| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | -| **Terveys** | ![Terveys](docs/screenshots/04-health.png) | -| **Kääntäjä** | ![Kääntäjä](docs/screenshots/05-translator.png) | -| **Asetukset** | ![Asetukset](docs/screenshots/06-settings.png) | -| **CLI-työkalut** | ![CLI-työkalut](docs/screenshots/07-cli-tools.png) | -| **Käyttölokit** | ![Käyttö](docs/screenshots/08-usage.png) | -| **Päätepisteet** | ![Päätepisteet](docs/screenshots/09-endpoint.png) |
| +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | + + --- ### 🤖 Free AI Provider for your favorite coding agents -_Yhdistä mikä tahansa tekoälyllä toimiva IDE- tai CLI-työkalu OmniRouten kautta – ilmainen API-yhdyskäytävä rajoittamattomaan koodaukseen._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ - +
OpenClaw
OpenClaw

- ⭐ 205 000 + ⭐ 205K
NanoBot
NanoBot

- ⭐ 20,9 kt + ⭐ 20.9K
PicoClaw
PicoClaw

- ⭐ 14,6 kt + ⭐ 14.6K
ZeroClaw
ZeroClaw

- ⭐ 9,9 kt + ⭐ 9.9K
IronClaw
IronClaw

- ⭐ 2,1 kt + ⭐ 2.1K
- Avoin koodi
+ OpenCode
OpenCode

- ⭐ 106 kt + ⭐ 106K
Codex CLI
Codex CLI

- ⭐ 60,8 kt + ⭐ 60.8K
Claude Code
Claude Code

- ⭐ 67,3 kt + ⭐ 67.3K
Gemini CLI
Gemini CLI

- ⭐ 94,7 kt + ⭐ 94.7K
- Kilokoodi
- Kilo-koodi + Kilo Code
+ Kilo Code

- ⭐ 15,5 tk + ⭐ 15.5K
-📡 Kaikki agentit muodostavat yhteyden http://localhost:20128/v1 tai http://cloud.omniroute.online/v1 kautta – yksi kokoonpano, rajattomasti malleja ja kiintiö--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Lopeta rahan tuhlaaminen ja rajojen ylittäminen:** +**Stop wasting money and hitting limits:** -- Tilauskiintiö vanhenee käyttämättä joka kuukausi -- Hintarajoitukset estävät sinua koodaamasta puolivälissä -- kalliita sovellusliittymiä (20-50 $/kk per tarjoaja) -- Manuaalinen vaihtaminen palveluntarjoajien välillä +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute ratkaisee tämän:** +**OmniRoute solves this:** -- ✅**Maksimoi tilaukset**- Seuraa kiintiötä, käytä jokainen bitti ennen nollausta -- ✅**Automaattinen palautus**- Tilaus → API-avain → Halpa → Ilmainen, nolla seisonta-aikaa -- ✅**Moni tili**- Pyöreä haku tilien välillä per palveluntarjoaja -- ✅**Universaali**- Toimii Claude Coden, Codexin, Gemini CLI:n, Cursorin, Clinen, OpenClawin ja minkä tahansa CLI-työkalun kanssa--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Liity yhteisöömme!**[WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Hanki apua, jaa vinkkejä ja pysy ajan tasalla. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Verkkosivusto**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Ongelmat**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [yhteisöryhmä](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Osallistuminen**: Katso [CONTRIBUTING.md](CONTRIBUTING.md), avaa PR tai valitse "hyvä ensimmäinen numero". -**Alkuperäinen projekti**: [9router by decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Kun avaat ongelman, suorita system-info-komento ja liitä luotu tiedosto:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Tämä luo `system-info.txt-tiedoston, joka sisältää Node.js-versiosi, OmniRoute-versiosi, käyttöjärjestelmän tiedot, asennetut CLI-työkalut (qoder, gemini, claude, codex, antigravity, droidi jne.), Docker/PM2-tilan ja järjestelmäpaketit – kaikki mitä tarvitsemme ongelmasi nopeaan toistamiseen. Liitä tiedosto suoraan GitHub-ongelmaasi.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Jokainen tekoälytyökaluja käyttävä kehittäjä kohtaa nämä ongelmat päivittäin.**OmniRoute luotiin ratkaisemaan ne kaikki – kustannusten ylityksistä alueellisiin lohkoihin, rikkinäisistä OAuth-virroista protokollatoimintoihin ja yrityksen havainnointikykyyn. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. - -💸 1. "Maksan kalliista tilauksesta, mutta silti rajoitukset häiritsevät minua" +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Kehittäjät maksavat 20–200 dollaria kuukaudessa Claude Prosta, Codex Prosta tai GitHub Copilotista. Maksamallakin kiintiöllä on katto – 5 tuntia käyttöä, viikkorajat tai minuuttirajoitukset. Koodausistunnon puolivälissä palveluntarjoaja lakkaa vastaamasta ja kehittäjä menettää virtauksen ja tuottavuuden. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Kuinka OmniRoute ratkaisee sen:** +**How OmniRoute solves it:** --**Smart 4-Tier Fallback**— Jos tilauskiintiö loppuu, ohjataan automaattisesti kohtaan API-avain → Halpa → Ilmainen ilman manuaalista toimenpiteitä --**Provider Limits Tracking**– Välimuistissa olevat kiintiön tilannevedokset päivittyvät palvelinpuolen aikataulun mukaan (oletus `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`), ja manuaalinen päivitys on saatavilla käyttöliittymässä --**Useiden tilien tuki**— Useita tilejä palveluntarjoajaa kohden automaattisella kierrätyksellä — kun yksi loppuu, vaihtuu seuraavaan --**Muokatut yhdistelmät**— Muokattavat varaketjut, joissa on 9 tasapainotusstrategiaa (prioriteetti, painotettu, täytä ensin, round-robin, P2C, satunnainen, vähiten käytetty, kustannusoptimoitu, tiukasti satunnainen) --**Codex Business Quotat**— Yritysten/Tiimien työtilan kiintiöiden valvonta suoraan kojelaudassa
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard - -🔌 2. "Minun täytyy käyttää useita palveluntarjoajia, mutta jokaisella on eri sovellusliittymä" + -OpenAI käyttää yhtä muotoa, Claude (Anthropic) käyttää toista, Gemini vielä toista. Jos kehittäjä haluaa testata eri palveluntarjoajien malleja tai vaihtoehtoja niiden välillä, hänen on määritettävä SDK:t uudelleen, muutettava päätepisteitä ja käsiteltävä yhteensopimattomia muotoja. Mukautetuilla palveluntarjoajilla (FriendLI, NIM) on mallista poikkeavat päätepisteet. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Kuinka OmniRoute ratkaisee sen:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Unified Endpoint**— Yksi "http://localhost:20128/v1" toimii välityspalvelimena kaikille yli 60 palveluntarjoajalle --**Format Translation**- Automaattinen ja läpinäkyvä: OpenAI ↔ Claude ↔ Gemini ↔ Responses API --**Response Sanitization**– Poistaa standardista poikkeavat kentät (`x_groq`, `usage_breakdown`, `service_tier`), jotka rikkovat OpenAI SDK v1.83+:n --**Roolin normalisointi**— Muuntaa "kehittäjä" → "järjestelmä" muille kuin OpenAI-palveluntarjoajille; "järjestelmä" → "käyttäjä" GLM:lle/ERNIE:lle --**Think Tag Extraction**– Purkaa ""-lohkot malleista, kuten DeepSeek R1, standardoituun "reasoning_content" -sisältöön --**Structured Output for Gemini**— `json_schema` → `responseMimeType`/`responseSchema` automaattinen muunnos --**"stream" oletusarvo on "false"**- yhdenmukaistuu OpenAI-spesifikaatioiden kanssa välttäen odottamattoman SSE:n Python/Rust/Go SDK:issa
+**How OmniRoute solves it:** - -🌐 3. "Tekoälypalveluntarjoajani estää alueeni/maani" +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Palveluntarjoajat, kuten OpenAI/Codex, estävät pääsyn tietyiltä maantieteellisiltä alueilta. Käyttäjät saavat virheitä, kuten "unsupported_country_region_territory" OAuth- ja API-yhteyksien aikana. Tämä on erityisen turhauttavaa kehitysmaiden kehittäjille. + -**Kuinka OmniRoute ratkaisee sen:** +
+🌐 3. "My AI provider blocks my region/country" --**3-tason välityspalvelimen määritys**– Muokattava välityspalvelin kolmella tasolla: yleinen (kaikki liikenne), palveluntarjoajakohtainen (vain yksi palveluntarjoaja) ja yhteys/avain --**Värikoodatut välityspalvelinmerkit**— Visuaaliset ilmaisimet: 🟢 maailmanlaajuinen välityspalvelin, 🟡 tarjoajan välityspalvelin, 🔵 yhteysvälityspalvelin, joka näyttää aina IP-osoitteen --**OAuth-tunnusten vaihto välityspalvelimen kautta**— OAuth-kulku kulkee myös välityspalvelimen kautta ja ratkaisee "unsupported_country_region_territory" --**Yhteystestit välityspalvelimen kautta**- Yhteystestit käyttävät määritettyä välityspalvelinta (ei enää suoraa ohitusta) --**SOCKS5-tuki**— Täysi SOCKS5-välityspalvelintuki lähtevään reititykseen --**TLS-sormenjälkien huijaus**— Selaimen kaltainen TLS-sormenjälki wreq-js:n kautta botin tunnistuksen ohittamiseksi --**🔏 CLI Fingerprint Matching**– Järjestää otsikot ja tekstikentät uudelleen vastaamaan alkuperäisiä CLI-binääriallekirjoituksia, mikä vähentää merkittävästi tilin ilmoittamisriskiä. Välityspalvelimen IP-osoite säilyy – saat sekä salaperäisen**- että**IP-peitetyksen samanaikaisesti
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. - -🆓 4. "Haluan käyttää tekoälyä koodaukseen, mutta minulla ei ole rahaa" +**How OmniRoute solves it:** -Kaikki eivät voi maksaa 20–200 dollaria kuukaudessa tekoälytilauksista. Opiskelijat, kehittäjät nousevista maista, harrastajat ja freelancerit tarvitsevat pääsyn laadukkaisiin malleihin ilman kustannuksia. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Kuinka OmniRoute ratkaisee sen:** + --**Free Tier Providers -sisäänrakennettu**— Natiivituki 100 % ilmaisille palveluntarjoajille: Qoder (5 rajoittamatonta mallia OAuthin kautta: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited3 mallia) qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID ilmaiseksi), Gemini CLI (180 000 tokenia / kuukausi ilmaiseksi) --**Ollama Cloud**— pilvissä isännöimät Ollama-mallit osoitteessa `api.ollama.com` ilmaisella "kevytkäyttö"-tasolla; käytä `ollamacloud/-etuliitettä --**Vain ilmaiset yhdistelmät**— Ketju `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = 0 $/kk ilman seisokkeja --**NVIDIA NIM Free Access**- ~40 RPM:n kehittäjä - ikuisesti ilmainen pääsy yli 70 malliin osoitteessa build.nvidia.com (siirrytään hyvityksistä puhtaisiin hintarajoihin) --**Kustannusoptimoitu strategia**— Reititysstrategia, joka valitsee automaattisesti halvimman saatavilla olevan palveluntarjoajan +
+🆓 4. "I want to use AI for coding but I have no money" - -🔒 5. "Minun täytyy suojata tekoälyyhdyskäytävääni luvattomalta käytöltä" +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -Kun paljastat tekoälyyhdyskäytävän verkkoon (LAN, VPS, Docker), kuka tahansa osoitteen tietävä voi kuluttaa kehittäjän tunnukset/kiintiöt. Ilman suojaa API:t ovat alttiita väärinkäytölle, nopealle injektiolle ja väärinkäytöksille. +**How OmniRoute solves it:** -**Kuinka OmniRoute ratkaisee sen:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**API-avainten hallinta**– Luominen, kierto ja laajuus palveluntarjoajan mukaan erillisellä /dashboard/api-manager-sivulla --**Mallitason käyttöoikeudet**– Rajoita API-avaimet tiettyihin malleihin ('openai/*', jokerimerkkimallit) Salli kaikki/Rajoita -kytkimellä --**Sovellusliittymän päätepistesuojaus**– vaadi avainta /v1/modelsille ja estä tietyt palveluntarjoajat luettelosta --**Auth Guard + CSRF-suojaus**— Kaikki kojelaudan reitit on suojattu "withAuth"-väliohjelmistolla + CSRF-tunnuksilla --**Rate Limiter**— IP-nopeuden rajoitus konfiguroitavilla ikkunoilla --**IP-suodatus**— Pääsynhallinnan sallittu-/estolista --**Prompt Injection Guard**— Desinfiointi haitallisia kehotusmalleja vastaan --**AES-256-GCM Encryption**— Tunnistetiedot on salattu lepotilassa
+ - -🛑 6. "Palvelajani kaatui ja menetin koodauskulkuni" +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -Tekoälypalveluntarjoajat voivat muuttua epävakaiksi, palauttaa 5xx-virheitä tai saavuttaa väliaikaiset nopeusrajoitukset. Jos kehittäjä on riippuvainen yhdestä palveluntarjoajasta, se keskeytyy. Ilman katkaisijoita toistuvat uudelleenyritykset voivat kaataa sovelluksen. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Kuinka OmniRoute ratkaisee sen:** +**How OmniRoute solves it:** --**Malleittainen katkaisija**- Automaattinen avautuminen/sulkeminen konfiguroitavilla kynnyksillä ja jäähdytys (suljettu/auki/puoliauki), mallikohtainen, jotta vältetään peräkkäiset lohkot --**Eksponentiaalinen peruutus**— Progressiiviset uudelleenyritysviiveet --**Anti-Thundering Herd**— Mutex + semaforisuoja samanaikaisia myrskyjä vastaan --**Yhdistelmävaraketjut**– Jos ensisijainen toimittaja epäonnistuu, putoaa automaattisesti ketjun läpi ilman väliintuloa --**Combo Circuit Breaker**— Poistaa automaattisesti käytöstä vialliset palveluntarjoajat yhdistelmäketjussa --**Health Dashboard**— käytettävyyden valvonta, katkaisijoiden tilat, lukitukset, välimuistitilastot, p50/p95/p99-viive
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest - -🔧 7. "Jokaisen tekoälytyökalun määrittäminen on työlästä ja toistuvaa" + -Kehittäjät käyttävät kursoria, Claude Codea, Codex CLI:tä, OpenClaw:ta, Gemini CLI:tä, Kilo Codea... Jokainen työkalu tarvitsee eri konfiguraation (API-päätepiste, avain, malli). Uudelleenmääritys toimittajaa tai mallia vaihdettaessa on ajanhukkaa. +
+🛑 6. "My provider went down and I lost my coding flow" -**Kuinka OmniRoute ratkaisee sen:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**CLI Tools Dashboard**- Erillinen sivu yhdellä napsautuksella Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline --**GitHub Copilot Config Generator**— Luo chatLanguageModels.json-tiedoston VS-koodille joukkomallin valinnalla --**Ohjattu käyttöönottotoiminto**— Ohjattu 4-vaiheinen asennus ensikertalaisille --**Yksi päätepiste, kaikki mallit**- Määritä "http://localhost:20128/v1" kerran, käytä yli 60 palveluntarjoajaa
+**How OmniRoute solves it:** - -🔑 8. "Useiden palveluntarjoajien OAuth-tunnusten hallinta on helvettiä" +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot – kaikki käyttävät OAuth 2.0:aa vanhentuvilla tunnuksilla. Kehittäjien täytyy todentaa jatkuvasti uudelleen, käsitellä "asiakassalaisuus puuttuu", "redirect_uri_mismatch" ja etäpalvelimien vikoja. OAuth LAN/VPS:ssä on erityisen ongelmallinen. + -**Kuinka OmniRoute ratkaisee sen:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Automaattinen tunnuksen päivitys**- OAuth-tunnukset päivittyvät taustalla ennen vanhenemista --**Sisäänrakennettu OAuth 2.0 (PKCE)**- Automaattinen kulku Claude Codelle, Codexille, Gemini CLI:lle, Copilotille, Kirolle, Qwenille, Qoderille --**Multi-Account OAuth**- Useita tilejä palveluntarjoajaa kohden JWT/ID-tunnuksen purkamisen kautta --**OAuth LAN/Remote Fix**- Yksityinen IP-tunnistus `redirect_uri':lle + manuaalinen URL-tila etäpalvelimille --**OAuth Nginxin takana**- Käyttää "window.location.origin" käänteisen välityspalvelimen yhteensopivuutta varten --**OAuth-etäopas**— Vaiheittainen opas Google Cloud -kirjautumistiedoille VPS/Dockerissa
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. - -📊 9. "En tiedä kuinka paljon kulutan tai minne" +**How OmniRoute solves it:** -Kehittäjät käyttävät useita maksullisia palveluntarjoajia, mutta heillä ei ole yhtenäistä näkemystä kuluttamisesta. Jokaisella palveluntarjoajalla on oma laskutuksen hallintapaneeli, mutta yhdistettyä näkymää ei ole. Odottamattomat kustannukset voivat kasaantua. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Kuinka OmniRoute ratkaisee sen:** + --**Cost Analytics Dashboard**– Token-kohtainen kustannusseuranta ja budjetin hallinta palveluntarjoajakohtaisesti --**Tasokohtaiset budjettirajat**– Tasokohtainen kulutuskatto, joka laukaisee automaattisen varauksen --**Malleittainen hinnoittelu**— Muokattavat hinnat mallikohtaisesti --**Käyttötilastot API-avainta kohti**— Pyyntömäärä ja viimeksi käytetty aikaleima avainta kohti --**Analytics Dashboard**- Tilastokortit, mallin käyttökaavio, toimittajataulukko onnistumisprosenteilla ja viiveellä +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" - -🐛 10. "En pysty diagnosoimaan tekoälypuhelujen virheitä ja ongelmia" +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Kun puhelu epäonnistuu, kehittäjä ei tiedä, oliko kyseessä nopeusrajoitus, vanhentunut tunnus, väärä muoto vai palveluntarjoajan virhe. Sirpaloituneet lokit eri terminaaleissa. Ilman havaittavuutta virheenkorjaus on yrityksen ja erehdysten menetelmää. +**How OmniRoute solves it:** -**Kuinka OmniRoute ratkaisee sen:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Yhdistettyjen lokien hallintapaneeli**- 4 välilehteä: pyyntölokit, välityspalvelimen lokit, tarkastuslokit, konsoli --**Console Log Viewer**- Reaaliaikainen päätetyylinen katseluohjelma värikoodatuilla tasoilla, automaattinen vieritys, haku, suodatin --**SQLite-välityspalvelimen lokit**— Pysyvät lokit, jotka kestävät palvelimen uudelleenkäynnistyksen --**Kääntäjän leikkikenttä**— 4 virheenkorjaustilaa: Playground (muodon käännös), Chat Tester (meno-paluu), testipenkki (erä), Live Monitor (reaaliaikainen) --**Pyyntötelemetria**— p50/p95/p99-latenssi + X-Request-Id-seuranta --**Tiedostopohjainen kirjaaminen rotaatiolla**– Sovelluslokit pyörivät koon, säilytyspäivien ja arkiston määrän mukaan; puhelulokin artefaktit kiertävät säilytyspäivien ja tiedostojen määrän mukaan --**Järjestelmätietoraportti**— `npm run system-info` luo `system-info.txt-tiedoston koko ympäristössäsi (solmuversio, OmniRoute-versio, käyttöjärjestelmä, CLI-työkalut, Docker/PM2-tila). Liitä se, kun ilmoitat ongelmista välittömässä triagessa.
+ - -🏗️ 11. "Yhdyskäytävän käyttöönotto ja ylläpito on monimutkaista" +
+📊 9. "I don't know how much I'm spending or where" -AI-välityspalvelimen asentaminen, määrittäminen ja ylläpito eri ympäristöissä (paikallinen, VPS, Docker, pilvi) on työvoimavaltaista. Ongelmat, kuten kovakoodatut polut, 'EACCES' hakemistoissa, porttiristiriidat ja cross-platform buildit lisäävät kitkaa. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Kuinka OmniRoute ratkaisee sen:** +**How OmniRoute solves it:** --**npm globaali asennus**— `npm install -g omniroute && omniroute` — valmis --**Docker Multi-Platform**- AMD64 + ARM64 natiivi (Apple Silicon, AWS Graviton, Raspberry Pi) --**Docker Compose -profiilit**— "base" (ei CLI-työkaluja) ja "cli" (Claude Coden, Codexin, OpenClawn kanssa) --**Electron Desktop App**- Natiivisovellus Windowsille/macOS:lle/Linuxille, jossa ilmaisinalue, automaattinen käynnistys, offline-tila --**Split-Port Mode**— API ja Dashboard erillisissä porteissa edistyneille skenaarioille (käänteinen välityspalvelin, konttiverkko) --**Cloud Sync**- Määritä synkronointi laitteiden välillä Cloudflare Workersin kautta --**DB-varmuuskopiot**— Kaikkien asetusten automaattinen varmuuskopiointi, palautus, vienti ja tuonti DISABLE_SQLITE_AUTO_BACKUP-toiminnolla ulkoisesti hallituille varmuuskopioille
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency - -🌍 12. "Käyttöliittymä on vain englanninkielinen ja tiimini ei puhu englantia" + -Ryhmät muissa kuin englanninkielisissä maissa, erityisesti Latinalaisessa Amerikassa, Aasiassa ja Euroopassa, kamppailevat vain englanninkielisten käyttöliittymien kanssa. Kielimuurit vähentävät käyttöönottoa ja lisäävät konfigurointivirheitä. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Kuinka OmniRoute ratkaisee sen:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Dashboard i18n — 30 kieltä**— Kaikki yli 500 näppäintä käännetty mukaan lukien arabia, bulgaria, tanska, saksa, espanja, suomi, ranska, heprea, hindi, unkari, indonesia, italia, japani, korea, malaiji, hollanti, norja, puola, portugali (PT/BR), romania, thai, venäjä, ukraina, slovakki, ruotsi, englanti --**RTL-tuki**— Tuki oikealta vasemmalle arabian ja heprean kielelle --**Multi-Language READMEs**- 30 täydellistä dokumentaation käännöstä --**Kielen valitsin**— Maapallokuvake otsikossa reaaliaikaista vaihtoa varten
+**How OmniRoute solves it:** - -🔄 13. "Tarvitsen muutakin kuin chatin – tarvitsen upotuksia, kuvia, ääntä" +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -Tekoäly ei ole vain chatin loppuun saattamista. Kehittäjien on luotava kuvia, litteroitava ääni, luotava upotuksia RAG:lle, järjestettävä asiakirjat uudelleen ja valvottava sisältöä. Jokaisella API:lla on eri päätepiste ja muoto. + -**Kuinka OmniRoute ratkaisee sen:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Upotukset**— `/v1/embeddings` kuudella palveluntarjoajalla ja 9+ mallilla --**Image Generation**— `/v1/images/generations` 10 tarjoajalla ja 20+ mallilla (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Tekstistä videoon**— `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) ja SD WebUI --**Text-to-Music**— `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) --**Äänitranskriptio**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Tekstistä puheeksi**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**ja olemassa olevat palveluntarjoajat --**Moderations**— `/v1/moderations` — Sisällön turvallisuustarkastukset --**Uudelleensijoitus**— `/v1/rerank` — Asiakirjan relevanssin uudelleensijoitus --**Responses API**- Täysi `/v1/responses` -tuki Codexille
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. - -🧪 14. "Minulla ei ole mahdollisuutta testata ja vertailla eri mallien laatua" +**How OmniRoute solves it:** -Kehittäjät haluavat tietää, mikä malli sopii parhaiten heidän käyttötapaukseensa – koodi, käännös, päättely – mutta manuaalinen vertailu on hidasta. Integroituja arviointityökaluja ei ole olemassa. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**Kuinka OmniRoute ratkaisee sen:** + --**LLM-arvioinnit**— Golden set -testaus 10 esiladatulla kotelolla, jotka kattavat tervehdyksen, matematiikan, maantieteen, koodin luomisen, JSON-yhteensopivuuden, käännöksen, merkinnän, turvallisuuden kieltämisen --**4 vastaavuusstrategiaa**— "tarkka", "sisältää", "säännöllinen lauseke", "muokattu" (JS-funktio) --**Translator Playground Test Bench**- Erätestaus useilla tuloilla ja odotetulla lähdöllä, tarjoajien välinen vertailu --**Chat Tester**- Täysi edestakainen matka visuaalisen vasteen renderöinnillä --**Live Monitor**— Reaaliaikainen tietovirta kaikista välityspalvelimen kautta kulkevista pyynnöistä +
+🌍 12. "The interface is English-only and my team doesn't speak English" - -📈 15. "Minun täytyy skaalata suorituskykyä menettämättä" +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -Pyynnön määrän kasvaessa samat kysymykset aiheuttavat päällekkäisiä kustannuksia välimuistiin tallentamatta. Ilman idempotenssia kaksoiskappaleet pyytävät jätteenkäsittelyä. Palveluntarjoajakohtaisia ​​hintarajoja on noudatettava. +**How OmniRoute solves it:** -**Kuinka OmniRoute ratkaisee sen:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Semanttinen välimuisti**– Kaksitasoinen välimuisti (allekirjoitus + semanttinen) vähentää kustannuksia ja viivettä --**Request Idempotency**— 5 sekunnin deduplikaatioikkuna identtisille pyynnöille --**nopeusrajoituksen tunnistus**– palveluntarjoajakohtainen RPM, pienin väli ja suurin samanaikainen seuranta --**Muokattavat nopeusrajoitukset**- Määritettävissä olevat oletusasetukset kohdassa Asetukset → Resilience with persistence --**API Key Validation Cache**– 3-tasoinen välimuisti tuotannon suorituskykyä varten --**Health Dashboard telemetrialla**- p50/p95/p99 latenssi, välimuistitilastot, käyttöaika
+ - -🤖 16. "Haluan hallita mallien käyttäytymistä maailmanlaajuisesti" +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Kehittäjät, jotka haluavat kaikki vastaukset tietyllä kielellä, tietyllä sävyllä tai haluavat rajoittaa perusteluita. Tämän määrittäminen jokaiseen työkaluun/pyyntöön on epäkäytännöllistä. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Kuinka OmniRoute ratkaisee sen:** +**How OmniRoute solves it:** --**Järjestelmäkehotteen lisäys**— Yleinen kehote koskee kaikkia pyyntöjä --**Thinking Budget Validation**– perustelutunnisteen allokoinnin ohjaus pyyntöä kohti (läpivienti, automaattinen, mukautettu, mukautuva) --**9 reititysstrategiaa**— Globaalit strategiat, jotka määrittävät pyyntöjen jakautumisen --**Wildcard Router**- "palveluntarjoaja/*" -mallit reitittävät dynaamisesti mille tahansa palveluntarjoajalle --**Yhdistelmä käyttöön/pois käytöstä**- Vaihda yhdistelmät suoraan kojelaudalta --**Provider Toggle**— Ota käyttöön tai poista käytöstä kaikki palveluntarjoajan yhteydet yhdellä napsautuksella --**Estetyt palveluntarjoajat**- Sulje pois tietyt palveluntarjoajat /v1/models-luettelosta
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex - -🧰 17. "Tarvitsen MCP-työkaluja ensiluokkaisina tuoteominaisuuksina" + -Monet tekoälyyhdyskäytävät paljastavat MCP:n vain piilotettuna toteutustietona. Tiimit tarvitsevat näkyvän, hallittavan toimintakerroksen. +
+🧪 14. "I have no way to test and compare quality across models" -**Kuinka OmniRoute ratkaisee sen:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP näkyy kojelaudan navigointi- ja päätepisteprotokolla-välilehdessä -- Erillinen MCP-hallintasivu, jossa on prosessit, työkalut, laajuudet ja tarkastus -- Sisäänrakennettu pikakäynnistys omniroute --mcp:lle ja asiakkaan käyttöönottoon
+**How OmniRoute solves it:** - -🧠 18. "Tarvitsen A2A-orkesterin synkronointi- ja stream-tehtäväpoluilla" +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Agenttityönkulut tarvitsevat sekä suoria vastauksia että pitkäkestoista suoratoistoa elinkaariohjauksella. + -**Kuinka OmniRoute ratkaisee sen:** +
+📈 15. "I need to scale without losing performance" -- A2A JSON-RPC -päätepiste ("POST /a2a") ja "message/send" ja "message/stream" -- SSE-suoratoisto päätetilan etenemisellä -- Tehtävien elinkaaren sovellusliittymät tehtäville/get- ja tasks/cancel
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. - -🛰️ 19. "Tarvitsen todellista MCP-prosessin kuntoa, en arvattua tilaa" +**How OmniRoute solves it:** -Operatiivisten tiimien on tiedettävä, onko MCP todella elossa, ei vain sitä, onko API tavoitettavissa. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**Kuinka OmniRoute ratkaisee sen:** + -- Ajonaikainen syketiedosto, jossa on PID, aikaleimat, kuljetus, työkalujen määrä ja laajuustila -- MCP-tilan API, joka yhdistää sykkeen + viimeaikaisen toiminnan -- Käyttöliittymän tilakortit prosessin / käytettävyyden / sydämenlyöntien tuoreudelle +
+🤖 16. "I want to control model behavior globally" - -📋 20. "Tarvitsen tarkastettavan MCP-työkalun suorittamisen" +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Kun työkalut muuttavat määrityksiä tai käynnistävät operaatioita, tiimit tarvitsevat rikosteknistä jäljitettävyyttä. +**How OmniRoute solves it:** -**Kuinka OmniRoute ratkaisee sen:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- SQLite-tuettu tarkastusloki MCP-työkalukutsuille -- Suodattimet työkalun, onnistumisen/epäonnistumisen, API-avaimen ja sivutuksen mukaan -- Kojelaudan tarkastustaulukko + tilastopäätepisteet automatisointia varten
+ - -🔐 21. "Tarvitsen laajennettuja MCP-oikeuksia integraatiota kohti" +
+🧰 17. "I need MCP tools as first-class product capabilities" -Eri asiakkailla tulisi olla vähiten käyttöoikeus työkaluluokkiin. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Kuinka OmniRoute ratkaisee sen:** +**How OmniRoute solves it:** -- 10 rakeista MCP-skooppia ohjattua työkalujen käyttöä varten -- Laajuuden valvonta ja näkyvyys MCP-hallintaliittymässä -- Turvallinen oletusasento käyttötyökaluille
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding - -⚙️ 22. "Tarvitsen toiminnanohjausta ilman uudelleensijoittamista" + -Tiimit tarvitsevat nopeita ajonaikaisia muutoksia tapausten tai kustannustapahtumien aikana. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**Kuinka OmniRoute ratkaisee sen:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Vaihda yhdistelmäaktivointia suoraan MCP-kojelaudalta -- Käytä joustavuusprofiileja ennalta määritetyistä käytäntöpaketeista -- Nollaa katkaisijan tila samasta käyttöpaneelista
+**How OmniRoute solves it:** - -🔄 23. "Tarvitsen live-A2A-tehtävän elinkaaren näkyvyyden ja peruutuksen" +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Ilman elinkaaren näkyvyyttä tehtäväkohtauksista tulee vaikeasti luokiteltuja. + -**Kuinka OmniRoute ratkaisee sen:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Tehtäväluettelo / suodatus tilan / taitojen mukaan ja sivutus -- Tehtävän metatietojen, tapahtumien ja artefaktien yksityiskohdat -- Tehtävän peruutuksen päätepiste ja käyttöliittymätoiminto vahvistuksen kanssa
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. - -🌊 24. "Tarvitsen aktiivisia suoratoistotietoja A2A-kuormitukseen" +**How OmniRoute solves it:** -Streaming-työnkulut edellyttävät toiminnallista tietoa samanaikaisuudesta ja reaaliaikaisista yhteyksistä. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**Kuinka OmniRoute ratkaisee sen:** + -- Aktiiviset virtalaskurit integroitu A2A-tilaan -- Viimeisen tehtävän aikaleima ja tilakohtaiset määrät -- A2A kojelautakortit reaaliaikaiseen toimintojen seurantaan +
+📋 20. "I need auditable MCP tool execution" - -🪪 25. "Tarvitsen asiakkaille tavallisen agenttihaun" +When tools mutate config or trigger ops actions, teams need forensic traceability. -Ulkoiset asiakkaat ja orkesterit tarvitsevat koneellisesti luettavaa metadataa käyttöönottoa varten. +**How OmniRoute solves it:** -**Kuinka OmniRoute ratkaisee sen:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- Agenttikortti esillä osoitteessa `/.well-known/agent.json' -- Johdon käyttöliittymässä näkyvät valmiudet ja taidot -- A2A status API sisältää etsintämetatiedot automatisointia varten
+ - -🧭 26. "Tarvitsen protokollan löydettävyyden tuotteen käyttökokemuksessa" +
+🔐 21. "I need scoped MCP permissions per integration" -Jos käyttäjät eivät löydä protokollapintoja, käyttöönoton ja tuen laatu heikkenee. +Different clients should have least-privilege access to tool categories. -**Kuinka OmniRoute ratkaisee sen:** +**How OmniRoute solves it:** -- Yhdistetty**Päätepisteet**-sivu, jossa on välilehdet välityspalvelin-, MCP-, A2A- ja API-päätepisteille -- Inline-palvelun tila vaihtuu (Online/Offline) MCP:lle ja A2A:lle -- Linkit yleiskatsauksesta erityisiin hallintavälilehtiin
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling - -🧪 27. "Tarvitsen päästä päähän -protokollan validoinnin oikeiden asiakkaiden kanssa" + -Valetestit eivät riitä vahvistamaan protokollan yhteensopivuutta ennen julkaisua. +
+⚙️ 22. "I need operational controls without redeploying" -**Kuinka OmniRoute ratkaisee sen:** +Teams need quick runtime changes during incidents or cost events. -- E2E-paketti, joka käynnistää sovelluksen ja käyttää todellista MCP SDK -asiakassiirtoa -- A2A-asiakas testaa virtojen löytämistä, lähettämistä, suoratoistoa, vastaanottamista ja peruuttamista -- Tarkista väitteet MCP-tarkastuksen ja A2A-tehtävien sovellusliittymien kanssa
+**How OmniRoute solves it:** - -📡 28. "Tarvitsen yhtenäisen havaittavuuden kaikissa käyttöliittymissä" +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -Havainnon jakaminen protokollan mukaan luo kuolleita kulmia ja pidemmän MTTR:n. + -**Kuinka OmniRoute ratkaisee sen:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Yhdistetyt kojelaudat/lokit/analytiikka yhdessä tuotteessa -- Terveys + auditointi + pyyntö telemetria OpenAI-, MCP- ja A2A-tasoilla -- Toiminnalliset sovellusliittymät tilaa ja automaatiota varten
+Without lifecycle visibility, task incidents become hard to triage. - -💼 29. "Tarvitsen yhden suoritusajan välityspalvelimelle + työkaluille + agentin orkestraatiolle" +**How OmniRoute solves it:** -Useiden erillisten palvelujen suorittaminen lisää käyttökustannuksia ja vikatiloja. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**Kuinka OmniRoute ratkaisee sen:** + -- OpenAI-yhteensopiva välityspalvelin, MCP-palvelin ja A2A-palvelin yhdessä pinossa -- Jaettu todennus, joustavuus, tietovarasto ja havaittavuus -- Yhdenmukainen toimintamalli kaikilla vuorovaikutuspinnoilla +
+🌊 24. "I need active stream metrics for A2A load" - -🚀 30. "Minun on lähetettävä agenttityönkulkuja ilman liimakoodin leviämistä" +Streaming workflows require operational insight into concurrency and live connections. -Tiimit menettävät nopeutta yhdistäessään useita ad-hoc-palveluita ja skriptejä. +**How OmniRoute solves it:** -**Kuinka OmniRoute ratkaisee sen:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Yhtenäinen päätepistestrategia asiakkaille ja edustajille -- Sisäänrakennetut protokollien hallinnan käyttöliittymät ja savun vahvistuspolut -- Tuotantovalmis perusta (turvallisuus, puunkorjuu, joustavuus, varmuuskopiointi)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Ohjekirja A: maksimoi maksullinen tilaus + halpa varmuuskopio**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -607,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Ohjekirja B: Nollahintainen koodauspino**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Playbook C: 24/7 aina päällä oleva varaketju**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -630,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Pelikirja D: Agentti toimii MCP:llä + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Määritä AI-koodaus minuuteissa hintaan**0 $/kk**. Yhdistä nämä ilmaiset tilit ja käytä sisäänrakennettua**Free Stack**-yhdistelmää. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Vaihe | Toiminta | Palveluntarjoajat avattu | -| ---- | --------------------------------------------------- | ------------------------------------------------------------------- | -| 1 | Yhdistä**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**rajaton**| -| 2 | Yhdistä**Qoder**(Google OAuth) | kimi-k2-ajattelu, qwen3-coder-plus, deepseek-r1... —**rajoittamaton**| -| 3 | Yhdistä**Qwen**(laitekoodi) | qwen3-coder-plus, qwen3-coder-flash... —**rajoittamaton**| -| 4 | Yhdistä**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180K/kk ilmaiseksi**| -| 5 | `/dashboard/combos` →**Ilmainen pino ($0)**malli | Round-robin kaikki ilmaiset palveluntarjoajat automaattisesti | +| Step | Action | Providers Unlocked | +| ---- | -------------------------------------------------- | ------------------------------------------------------------------ | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Osoita mikä tahansa IDE/CLI osoitteeseen:**"http://localhost:20128/v1" · API-avain: "any-string" · Valmis. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Valinnainen lisäkattavuus (myös ilmainen):**Groq API-avain (30 RPM ilmaiseksi), NVIDIA NIM (40 RPM ilmaiseksi, 70+ mallit), Cerebras (1 milj. tok/päivä), LongCat API-avain (50 milj. tokenia/päivä!), Cloudflare Workers AI (10 000 neuronia/vrk, 50+ mallia).## Pikakäynnistys +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Pikakäynnistys ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **pnpm-käyttäjät:**Suorita `pnpm approve-builds -g` asennuksen jälkeen, jotta voit ottaa käyttöön `better-sqlite3` ja `@swc/core` vaatimat alkuperäiset koontiskriptit: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > > ```bash > pnpm install -g omniroute -> pnpm approve-builds -g # Valitse kaikki paketit → hyväksy -> kaikkialla +> pnpm approve-builds -g # Select all packages → approve +> omniroute > ``` -Hallintapaneeli avautuu osoitteessa "http://localhost:20128" ja API-perus-URL-osoite on "http://localhost:20128/v1". +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Komento | Kuvaus | -| ----------------------- | -------------------------------------------------------------------- | -| `omniroute` | Käynnistä palvelin (`PORT=20128`, API ja kojelauta samassa portissa) | -| `omniroute --port 3000` | Aseta kanoninen/API-portiksi 3000 | -| `omniroute --mcp` | Käynnistä MCP-palvelin (stdio-kuljetus) | -| `omniroute --no-open` | Älä avaa selainta automaattisesti | -| `omniroute --help` | Näytä ohje | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Valinnainen jaettu porttitila:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -Useimpiin käyttöönottoihin tarvitset vain: +For most deployments, you only need: -| Muuttuja | Oletus | Tarkoitus | -| ------------------------- | ------------------------------ | -------------------------------------------------------------- --------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | "600000" | Jaettu lähtötaso ylävirran noutoa, piilotettuja Undici-aikakatkaisuja, TLS-sormenjälkipyyntöjä ja API-siltapyyntöjen/välityspalvelinten aikakatkaisuja varten | -| `STREAM_IDLE_TIMEOUT_MS` | perii REQUEST_TIMEOUT_MS | Suurin väli suoratoistopalojen välillä ennen kuin OmniRoute keskeyttää SSE-virran | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -Taaksepäin yhteensopivuus säilyy: olemassa olevat "FETCH_TIMEOUT_MS", "API_BRIDGE_PROXY_TIMEOUT_MS" ja muut tasokohtaiset aikakatkaisumuuttujat toimivat edelleen ja ohittavat jaetun perustason. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Edistyneet ohitukset ovat käytettävissä, jos tarvitset tarkempaa ohjausta:| Muuttuja | Oletus | Tarkoitus | -| ----------------------------------------- | ------------------------------------------- | --------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | perii REQUEST_TIMEOUT_MS | Päähaun keskeytyssignaalin käyttämä ylävirran pyynnön kokonaisaikakatkaisu | -| `FETCH_HEADERS_TIMEOUT_MS` | perii `FETCH_TIMEOUT_MS` | Undici-aikaraja ylävirran vastausotsikoiden vastaanottamiselle | -| `FETCH_BODY_TIMEOUT_MS` | perii `FETCH_TIMEOUT_MS` | Undici-aikaraja ylävirran runkokappaleiden välillä (`0` poistaa sen käytöstä) | -| `FETCH_CONNECT_TIMEOUT_MS` | "30000" | Undici TCP-yhteyden aikakatkaisu | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | "4000" | Undici idle keep-alive socket timeout | -| `TLS_CLIENT_TIMEOUT_MS` | perii `FETCH_TIMEOUT_MS` | Aikakatkaisu wreq-js:n kautta tehdyille TLS-sormenjälkipyynnöille | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | perii REQUEST_TIMEOUT_MS tai 30000 | Aikakatkaisu välityspalvelimen `/v1' edelleenlähetykselle API-portista kojelautaporttiin | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | "max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)" | Saapuvan pyynnön aikakatkaisu API-siltapalvelimella | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | "60000" | Saapuvan otsikon aikakatkaisu API-siltapalvelimella | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | "5000" | Keep-alive aikakatkaisu API-siltapalvelimella | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | "0" | Socketin passiivisuuden aikakatkaisu API-siltapalvelimessa (`0` poistaa sen käytöstä) | +Advanced overrides are available if you need finer control: -Jos käytät OmniRoutea Nginxin, Caddyn, Cloudflaren tai muun käänteisen välityspalvelimen takana, varmista, että välityspalvelin -aikakatkaisut ovat myös korkeammat kuin OmniRoute-streamin/haun aikakatkaisut.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Avaa Dashboard → Providers ja yhdistä vähintään yksi palveluntarjoaja (OAuth- tai API-avain). -2. Avaa Dashboard → `Endpoints` ja luo API-avain. -3. (Valinnainen) Avaa Dashboard → "Yhdistelmät" ja aseta varaketju.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Toimii Claude Coden, Codex CLI:n, Gemini CLI:n, Cursorin, Clinen, OpenClawn, OpenCoden ja OpenAI-yhteensopivien SDK:iden kanssa.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (työkaluohjattuihin toimintoihin):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` +Then connect your MCP client over `stdio` and test tools like: -Yhdistä sitten MCP-asiakkaasi "stdion" kautta ja testaa työkaluja, kuten: +- `omniroute_get_health` +- `omniroute_list_combos` -- `omniroute_get_health' -- "omniroute_list_combos". +**A2A (for agent-to-agent workflows):** -**A2A (agenttien välisille työnkuluille):**```bash +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -759,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Tämä sarja tarkistaa todelliset MCP- ja A2A-asiakasvirrat käynnissä olevaa sovellusta vastaan.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -767,13 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` - -Vid Linux (`xbps-src` malli) +
+Void Linux (`xbps-src` template) -Void Linux -käyttäjille voit rakentaa alkuperäisen paketin käyttämällä `xbps-src`. Tallenna tämä lohko nimellä "srcpkgs/omniroute/template":```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -785,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -793,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -869,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -880,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute on saatavilla julkisena Docker-kuvana [Docker Hubista](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Pikaajo:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -890,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**Ympäristötiedostolla:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Docker Composen käyttäminen:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Dashboard-tuki Dockerin käyttöönottoille sisältää nyt yhden napsautuksen**Cloudflare Quick Tunnel**kohdassa "Dashboard → Endpoints". Ensimmäinen sallii lataukset "cloudflared" vain tarvittaessa, käynnistää väliaikaisen tunnelin nykyiseen "/v1"-päätepisteeseen ja näyttää luodun "https://\*.trycloudflare.com/v1" URL-osoitteen suoraan tavallisen julkisen URL-osoitteesi alapuolelle. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Huomautuksia: +Notes: -- Quick Tunnelin URL-osoitteet ovat väliaikaisia ja muuttuvat jokaisen uudelleenkäynnistyksen jälkeen. -- Pikatunneleita ei palauteta automaattisesti OmniRouten tai kontin uudelleenkäynnistyksen jälkeen. Ota ne uudelleen käyttöön kojelaudasta tarvittaessa. -- Hallittu asennus tukee tällä hetkellä Linuxia, macOS:ää ja Windowsia x64-/arm64-käyttöjärjestelmässä. -- Hallitut pikatunnelit käyttävät oletusarvoisesti HTTP/2-siirtoa, jotta vältetään meluisat QUIC UDP -puskurivaroitukset rajoitetuissa säiliöympäristöissä. Aseta CLOUDFLARED_PROTOCOL=quic tai auto, jos haluat toisenlaisen kuljetuksen. -- Docker-kuvat niputtavat järjestelmän CA-juuret ja välittävät ne hallitulle "cloudflaredille", mikä välttää TLS-luottamushäiriöt, kun tunneli käynnistyy säilön sisällä. -- SQLite toimii WAL-tilassa. `docker stop`:n tulee antaa päättyä, jotta OmniRoute voi tarkistaa viimeisimmät muutokset takaisin `storage.sqlite`-tiedostoon. -- Mukana olevissa Compose-tiedostoissa on jo asetettu 40 s stop lisäaika. Jos suoritat kuvan suoraan, pidä `--stop-timeout 40` (tai vastaava), jotta manuaaliset pysäytykset eivät katkaise sammutusta. -- Aseta `CLOUDFLARED_BIN=/absolute/path/to/cloudflared', jos haluat OmniRouten käyttävän olemassa olevaa binaaria sen lataamisen sijaan. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Docker Compose with Caddy (HTTPS Auto-TLS) käyttäminen:** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute voidaan paljastaa turvallisesti käyttämällä Caddyn automaattista SSL-tilausta. Varmista, että verkkotunnuksesi DNS-tietue A osoittaa palvelimesi IP-osoitteeseen.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` +| Image | Tag | Size | Description | +| ------------------------ | -------- | ------ | --------------------- | +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | -| Kuva | Tag | Koko | Kuvaus | -| ------------------------- | -------- | ------ | ---------------------- | -| "diegosouzapw/omniroute" | "viimeisin" | ~250 Mt | Uusin vakaa julkaisu | -| "diegosouzapw/omniroute" | "1.0.3" | ~250 Mt | Nykyinen versio |--- +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**UUSI!**OmniRoute on nyt saatavilla**alkuperäisenä työpöytäsovelluksena**Windowsille, macOS:lle ja Linuxille. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Suorita OmniRoute itsenäisenä työpöytäsovelluksena – ei päätelaitetta, ei selainta, ei vaadi Internetiä paikallisiin malleihin. Elektronipohjainen sovellus sisältää: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Native Window**- Oma sovellusikkuna, jossa on integraatio järjestelmälokeroon -- 🔄**Automaattinen käynnistys**— Käynnistä OmniRoute järjestelmään kirjautuessasi -- 🔔**Alkuperäiset ilmoitukset**- Saat ilmoituksia kiintiön loppumisesta tai palveluntarjoajan ongelmista -- ⚡**Asennus yhdellä napsautuksella**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Offline-tila**— Toimii täysin offline-tilassa mukana toimitetun palvelimen kanssa### Pikakäynnistys +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Pikakäynnistys ```bash # Development mode @@ -979,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -Kun OmniRoute on minimoitu, se elää ilmaisinalueellasi nopeilla toimilla: +When minimized, OmniRoute lives in your system tray with quick actions: -- Avaa kojelauta -- Vaihda palvelimen portti -- Lopeta sovellus +- Open dashboard +- Change server port +- Quit application -📖 Täydellinen dokumentaatio: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Taso | Palveluntarjoaja | Kustannukset | Kiintiön nollaus | Paras | -| ---------------- | --------------------------- | ---------------------------------- | ---------------------- | ---------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -| **💳 TILAUS** | Claude Code (Pro) | 20 dollaria/kk | 5h + viikoittain | jo tilattu | -| | Codex (Plus/Pro) | 20-200 $/kk | 5h + viikoittain | OpenAI-käyttäjät | -| | Gemini CLI | **ILMAINEN** | 180 tk/kk + 1 tk/päivä | Kaikki! | -| | GitHub Copilot | 10-19 $/kk | Kuukausittain | GitHub-käyttäjät | -| **🔑 API-AVAIN** | NVIDIA NIM | **ILMAINEN**(kehittäjä ikuisesti) | ~40 rpm | 70+ avointa mallia | -| | Aivot | **ILMAINEN**(1M tok/päivä) | 60K TPM / 30 RPM | Maailman nopein | -| | Groq | **ILMAINEN**(30 RPM) | 14.4K RPD | Erittäin nopea Llama/Gemma | -| | DeepSeek V3.2 | 0,27 $/1,10 $ per 1 milj | Ei yhtään | Paras hinta/laatu perustelu | -| | xAI Grok-4 Fast | **0,20 $/0,50 $ per 1 milj**🆕 | Ei yhtään | Nopein + työkalukutsu, ultralow | -| | xAI Grok-4 (vakio) | 0,20 $/1,50 $ per 1 milj 🆕 | Ei yhtään | Päättelyn lippulaiva xAI:lta | -| | Mistral | Ilmainen kokeilu + maksullinen | Hinta rajoitettu | Eurooppalainen tekoäly | -| | OpenRouter | Maksu per käyttö | Ei yhtään | 100+ mallia agr. | -| **💰 EDULLISET** | GLM-5 (Z.AI:n kautta) 🆕 | 0,5 $/1 milj. | Päivittäin klo 10 | 128K lähtö, uusin lippulaiva | -| | GLM-4.7 | 0,6 $/1 milj. | Päivittäin klo 10 | Budjetin varmuuskopio | -| | MiniMax M2.5 🆕 | 0,3 $/1 milj. tulo | 5 tunnin rullaus | Päättely + agenttitehtävät | -| | MiniMax M2.1 | 0,2 $/1 milj. | 5 tunnin rullaus | Halvin vaihtoehto | -| | Kimi K2.5 (Moonshot API) 🆕 | Maksu per käyttö | Ei yhtään | Suora Moonshot API -käyttö | -| | Kimi K2 | 9 dollaria/kk asunto | 10 milj. rahakkeita/kk | Ennustettavat kustannukset | -| **🆓 ILMAINEN** | Qoder | **0 $** | Rajoittamaton | 5 mallia rajoittamaton | -| | Qwen | **0 $** | Rajoittamaton | 4 mallia rajoittamaton | -| | Kiro | **0 $** | Rajoittamaton | Claude Sonnet/Haiku (AWS Builder) | -| | LongCat Flash-Lite 🆕 | **0 $**(50 milj. tok/päivä 🔥) | 1 RPS | Suurin ilmainen kiintiö maailmassa | -| | Pölytys AI 🆕 | **0 $**(avainta ei tarvita) | 1 kpl/15 s | GPT-5, Claude, DeepSeek, Llama 4 | -| | Cloudflare Workers AI 🆕 | **0 $**(10 000 neuronia/päivä) | ~150 tk/päivä | Yli 50 mallia, globaali reuna | -| | Scaleway AI 🆕 | **0 $**(yhteensä 1 milj. tokeneja) | Hinta rajoitettu | EU/GDPR, Qwen3 235B, Llama 70B | > 🆕**Uusia malleja lisätty (maaliskuu 2026):**Grok-4 Fast -perhe hintaan 0,20 $/0,50 $/M (vertailuarvo 1143 ms – 30 % nopeampi kuin Gemini 2.5 Flash), GLM-5 Z.AI:n kautta 128K:n lähdöllä, MiniMax M2.5 Vc3 -perustelu, KiepSeed2-päivitys. Moonshot Direct API. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 0 dollarin yhdistelmäpino – täydellinen ilmainen asennus:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Nolla kustannuksia. Ei koskaan lopeta koodausta.**Määritä tämä yhdeksi OmniRoute-yhdistelmäksi, ja kaikki palautukset tapahtuvat automaattisesti – ei manuaalista vaihtoa koskaan.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Kaikki alla olevat mallit ovat**100 % ilmaisia ​​ilman luottokorttia**. OmniRoute reitittää automaattisesti niiden välillä, kun yksi kiintiö loppuu – yhdistä ne kaikki rikkomattomaksi 0 dollarin yhdistelmäksi.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Malli | Etuliite | Raja | Hintarajoitus | -| -------------------- | ------ | ------------- | ---------------------- | -| `claude-sonnet-4,5` | `kr/` |**Rajoittamaton**| Ei raportoitu päivittäistä ylärajaa | -| `claude-haiku-4,5` | `kr/` |**Rajoittamaton**| Ei raportoitu päivittäistä ylärajaa | -| `claude-opus-4.6` | `kr/` |**Rajoittamaton**| Uusin Opus kautta Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) -| Malli | Etuliite | Raja | Hintarajoitus | -| ------------------- | ------ | ------------- | ---------------- | -| `kimi-k2-ajattelu` | "jos/" |**Rajoittamaton**| Ei ilmoitettu yläraja | -| "qwen3-coder-plus" | "jos/" |**Rajoittamaton**| Ei ilmoitettu yläraja | -| `deepseek-r1` | "jos/" |**Rajoittamaton**| Ei ilmoitettu yläraja | -| "minimi-m2,1" | "jos/" |**Rajoittamaton**| Ei ilmoitettu yläraja | -| "kimi-k2" | "jos/" |**Rajoittamaton**| Ei ilmoitettu yläraja | +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | --------------------- | +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -> Suositeltu yhteystapa:**Personal Access Token + `qodercli`**. Selain OAuth on -> kokeellinen ja oletuksena poistettu käytöstä, ellei QODER_OAUTH_*-ympäristömuuttujia ole määritetty.### 🟡 QWEN MODELS (Device Code Auth) +### 🟢 QODER MODELS (Free PAT via qodercli) -| Malli | Etuliite | Raja | Hintarajoitus | -| -------------------- | ------ | ------------- | -------------------- | -| "qwen3-coder-plus" | `qw/` |**Rajoittamaton**| Ei ilmoitettu yläraja | -| "qwen3-coder-flash" | `qw/` |**Rajoittamaton**| Ei ilmoitettu yläraja | -| "qwen3-coder-next" | `qw/` |**Rajoittamaton**| Ei ilmoitettu yläraja | -| "näön malli" | `qw/` |**Rajoittamaton**| Multimodaalinen (kuvat) |### 🟣 GEMINI CLI (Google OAuth) +| Model | Prefix | Limit | Rate Limit | +| ------------------ | ------ | ------------- | --------------- | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -| Malli | Etuliite | Raja | Hintarajoitus | -| ------------------------- | ------ | ---------------------------- | ------------- | -| `gemini-3-flash-preview` | `gc/` |**180 tk/kk**+ 1 tk/päivä | Kuukausittainen nollaus | -| "gemini-2.5-pro" | `gc/` | 180 000/kk (jaettu pool) | Korkea laatu |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| Taso | Päiväraja | Hintarajoitus | Huomautuksia | -| ----------- | ------------ | ----------- | ------------------------------------------------------- | -| Ilmainen (kehittäjä) | Ei tunnuskorkkia |**~40 RPM**| 70+ mallia; siirtyminen puhtaisiin korkorajoihin vuoden 2025 puolivälissä | +### 🟡 QWEN MODELS (Device Code Auth) -Suositut ilmaiset mallit: moonshotai/kimi-k2.5 (Kimi K2.5), z-ai/glm4.7 (GLM 4.7), deepseek-ai/deepseek-v3.2 (DeepSeek V3.2), nvidia/llama-3.3-70b-deepseekr,/deepseekr### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | ------------------- | +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Taso | Päiväraja | Hintarajoitus | Huomautuksia | -| ---- | ------------------ | ----------------- | -------------------------------------------- | -| Ilmainen |**1 milj. rahakkeita/päivä**| 60K TPM / 30 RPM | Maailman nopein LLM-päätelmä; nollautuu päivittäin | +### 🟣 GEMINI CLI (Google OAuth) -Saatavilla ilmaiseksi: "llama-3.3-70b", "llama-3.1-8b", "deepseek-r1-distill-llama-70b"### 🔴 GROQ (Free API Key — console.groq.com) +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -| Taso | Päiväraja | Hintarajoitus | Huomautuksia | -| ---- | ------------- | ----------------- | ------------------------------------------ | -| Ilmainen |**14.4K RPD**| 30 rpm mallia kohden | Ei luottokorttia; 429 rajalla, ei veloiteta | +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) -Saatavilla ilmaiseksi: "llama-3.3-70b-versatile", "gemma2-9b-it", "mixtral-8x7b", "whisper-large-v3"### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +| Tier | Daily Limit | Rate Limit | Notes | +| ---------- | ------------ | ----------- | ------------------------------------------------------ | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -| Malli | Etuliite | Päivittäinen ilmainen kiintiö | Huomautuksia | -| ------------------------------ | ------ | ------------------ | ------------------------ | -| "LongCat-Flash-Lite" | `lc/` |**50 milj. rahakkeita**💥 | Suurin ilmainen kiintiö koskaan | -| "LongCat-Flash-Chat" | `lc/` | 500 000 kuponkia | Monipuolinen chat | -| "LongCat-Flash-Thinking" | `lc/` | 500 000 kuponkia | Päättely / CoT | -| "LongCat-Flash-Thinking-2601" | `lc/` | 500 000 kuponkia | Tammikuun 2026 versio | -| "LongCat-Flash-Omni-2603" | `lc/` | 500 000 kuponkia | Multimodaalinen | +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -> 100 % ilmainen julkisessa beta-tilassa. Rekisteröidy osoitteessa [longcat.chat](https://longcat.chat) sähköpostitse tai puhelimitse. Nollautuu päivittäin 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -| Malli | Etuliite | Hintarajoitus | Toimittaja Takana | -| ----------- | ------ | ----------- | ------------------- | -| "openai" | `pol/` | 1 tarve/15 s | GPT-5 | -| `claude` | `pol/` | 1 tarve/15 s | Antrooppinen Claude | -| "kaksoset" | `pol/` | 1 tarve/15 s | Google Gemini | -| `deepseek` | `pol/` | 1 kpl/15 s | DeepSeek V3 | -| `laama` | `pol/` | 1 kpl/15 s | Meta Llama 4 Scout | -| "mistral" | `pol/` | 1 kpl/15 s | Mistral AI | +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -> ✨**Nolla kitkaa:**Ei rekisteröitymistä, ei API-avainta. Lisää Pollinations-palveluntarjoaja tyhjällä avainkentällä ja se toimii välittömästi.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -| Taso | Päivittäiset neuronit | Vastaava käyttö | Huomautuksia | -| ---- | ------------- | ---------------------------------------- | ------------------------ | -| Ilmainen |**10 000**| ~150 LLM vastinetta / 500 s ääni / 15 000 upotusta | Maailmanlaajuinen etu, yli 50 mallia | +### 🔴 GROQ (Free API Key — console.groq.com) -Suositut ilmaiset mallit: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (ilmainen ääni!), `@cf/qwen/qwen2.5-coder-`15b-coder- +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ------------- | ---------------- | ----------------------------------------- | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -> Vaatii API-tunnuksen + tilitunnuksen osoitteesta [dash.cloudflare.com](https://dash.cloudflare.com). Tallenna tilitunnus palveluntarjoajan asetuksiin.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -| Taso | Ilmainen kiintiö | Sijainti | Huomautuksia | -| ---- | ------------- | ------------ | ------------------------------------ | -| Ilmainen |**1 milj. rahakkeita**| 🇫🇷 Pariisi, EU | Luottokorttia ei tarvita rajoissa | +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 -Saatavilla ilmaiseksi: "qwen3-235b-a22b-instruct-2507" (Qwen3 235B!), "llama-3.1-70b-instruct", "mistral-small-3.2-24b-instruct-2506", "deepseek-v3-032" +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | -> EU/GDPR-yhteensopiva. Hanki API-avain osoitteesta [console.scaleway.com](https://console.scaleway.com). +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. ->**💡 The Ultimate Free Stack (11 tarjoajaa, 0 dollaria ikuisesti):** +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | +| ---------- | ------ | ---------- | ------------------ | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | + +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. + +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 + +| Tier | Daily Neurons | Equivalent Usage | Notes | +| ---- | ------------- | --------------------------------------- | ----------------------- | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | + +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` + +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. + +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | +| ---- | ------------- | ------------ | ----------------------------------- | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | + +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` + +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). + +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -> Qoder (if/) → kimi-k2-ajattelu, qwen3-coder-plus, deepseek-r1 UNLIMITED -> LongCat Lite (lc/) → LongCat-Flash-Lite – 50 miljoonaa rahaketta/päivä 🔥 -> Pölytys (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — avainta ei tarvita -> Qwen (qw/) → qwen3-kooderimallit RAJOITTAMATTA -> Gemini (gemini/) → Gemini 2.5 Flash – 1 500 rekv/päivä ilmaiseksi -> Cloudflare AI (vrt./) → 50+ mallia – 10 000 neuronia/päivä -> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1 miljoona ilmaista rahaketta (EU) -> Groq (groq/) → Llama/Gemma – 14,4 000 req/päivä erittäin nopea -> NVIDIA NIM (nvidia/) → 70+ avointa mallia – 40 RPM ikuisesti -> Aivot (cerebras/) → Llama/Qwen maailman nopein – 1 milj. tok/päivä -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Literoi mikä tahansa ääni/video hintaan**$0**— Deepgram-johdot 200 dollarilla ilmaiseksi, AssemblyAI 50 dollarin varavara, Groq Whisper rajoittamattomana hätävarmuuskopiona. +## 🎙️ Free Transcription Combo -| Palveluntarjoaja | Ilmaisia ​​luottoja | Paras malli | Hintarajoitus | -| ------------------ | ----------------------- | --------------------------------------------- | ----------------------------- | -| 🟢**Deepgram**|**200 dollaria ilmaiseksi**(kirjautuminen) | `nova-3` — paras tarkkuus, yli 30 kieltä | Ei RPM-rajoitusta ilmaisille luottoille | -| 🔵**AssemblyAI**|**50 dollaria ilmaiseksi**(kirjautuminen) | "universal-3-pro" — luvut, tunnelma, henkilötiedot | Ei RPM-rajoitusta ilmaisille luottoille | -| 🔴**Groq**|**Ilmainen ikuisesti**| `whisper-large-v3` — OpenAI Whisper | 30 RPM (nopeus rajoitettu) | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. -**Ehdotettu yhdistelmä kohdassa `/dashboard/combos':**``` +| Provider | Free Credits | Best Model | Rate Limit | +| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | + +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Sitten `/dashboard/media` →**Transkriptio**-välilehti: lataa mikä tahansa ääni- tai videotiedosto → valitse yhdistelmäpäätepiste → hanki transkriptio tuetuissa muodoissa.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 on rakennettu toiminnalliseksi alustaksi, ei vain välityspalvelimeksi.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Ominaisuus | Mitä se tekee | -| ------------------------------------------------ | ----------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Grok-4 Fast Family** | xAI-mallit hintaan 0,20 $/0,50 $/M – vertailuarvo 1143 ms (30 % nopeampi kuin Gemini 2.5 Flash) | -| 🧠**GLM-5 Z.AI:n kautta** | 128 000 tuloskonteksti, 0,5 $/1 milj. GLM-perheen uusin lippulaiva | -| 🔮**MiniMax M2.5** | Päättely + agenttitehtävät hintaan 0,30 $/1M – merkittävä parannus M2.1:stä | -| 🎯**ToolCalling Flag mallikohtainen** | Mallikohtainen työkalukutsu: tosi/epätosi rekisterissä — AutoCombo ohittaa mallit, joissa ei ole työkaluja | -| 🌍**Monikielinen tarkoituksentunnistus** | PT/ZH/ES/AR avainsanat AutoCombo-pisteytyksissä — parempi mallivalinta ei-englanninkieliselle sisällölle | -| 📊**Vertailuarvoihin perustuvat varaehdotukset** | Todellinen p95-viive live-pyyntöjen syötteiden yhdistelmäpisteistä — AutoCombo oppii todellisista tiedoista | -| 🔁**Pyydä päällekkäisyyden poistamista** | Sisältö-hash-pohjainen dedup-ikkuna — turvallinen usealle agentille, estää päällekkäiset veloitukset | -| 🔌**Pluggable RouterStrategy** | Laajentuva RouterStrategy-käyttöliittymä — lisää mukautettu reitityslogiikka laajennuksiksi | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Ominaisuus | Mitä se tekee | -| ------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Model Playground** | Dashboard-sivu minkä tahansa mallin testaamiseen suoraan – palveluntarjoajan/mallin/päätepisteen valitsimet, Monaco Editor, suoratoisto, keskeytys, ajoitus | -| 🔏**CLI-sormenjälkien vastaavuus** | Palveluntarjoajakohtainen otsikko/runkojärjestys vastaamaan alkuperäisiä CLI-allekirjoituksia – vaihda palveluntarjoajan mukaan kohdassa Asetukset > Suojaus.**Välipalvelimesi IP-osoite säilyy** | -| 🤝**ACP-tuki (Agent Client Protocol)** | CLI-agentin etsintä (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 muuta), prosessin synnyttäjä, /api/acp/agents-päätepiste | -| 🤖**ACP Agents Dashboard** | Vianetsintä › Agentit -sivu — 14 agentin ruudukko, jossa on asennustila, versio ja mukautettu agenttilomake mille tahansa CLI-työkalulle.**OpenCode**-käyttäjät saavat "Download opencode.json" -painikkeen, joka luo automaattisesti käyttövalmiin kokoonpanon kaikille saatavilla oleville malleille. | -| 🔧**Muokatun mallin `apiFormat` -reititys** | Mukautetut mallit, joissa on `apiFormat: "responses"`, ohjaavat nyt oikein Responses API -kääntäjään | -| 🏢**Codex Workspace Isolation** | Useita Codex-työtiloja sähköpostissa — OAuth erottaa yhteydet oikein työtilan tunnuksen | -| 🔄**Automaattinen elektroninen päivitys** | Työpöytäsovellus tarkistaa päivitykset + automaattinen asennus uudelleenkäynnistyksen yhteydessä | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Ominaisuus | Mitä se tekee | -| ------------------------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**MCP-palvelin (25 työkalua)** | IDE/agenttityökalut kolmen kuljetuksen kautta: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 ydintä + 3 muistia + 4 taitotyökalua | -| 🤝**A2A-palvelin (JSON-RPC + SSE)** | Agenttien välinen tehtävien suorittaminen synkronointi- ja suoratoistovirroilla | -| 🧭**Konsolidoitu päätepistesivu** | Välilehdillä varustettu hallintasivu Endpoint Proxy-, MCP-, A2A- ja API Endpoints -välilehdillä | -| 🎚️**Palvelun käyttöönoton/poistamisen valinnat** | ON/OFF kytkimet MCP:lle ja A2A:lle ja asetusten pysyvyys (oletus: OFF) | -| 🛰️**MCP Runtime Heartbeat** | Todellinen prosessin tila (pid, käytettävyys, sykeikä, kuljetus, mittaustila) | -| 📋**MCP Audit Trail** | Suodatettavat tarkastuslokit onnistumis/epäonnistuminen ja avainattribuutio | -| 🔐**MCP Scope Enforcement** | 10 yksityiskohtaista käyttöoikeutta työkalujen hallitukselle | -| 📡**A2A-tehtävän elinkaaren hallinta** | Listaa/suodata tehtäviä, tarkasta tapahtumat/artefaktit, peruuta käynnissä olevat tehtävät | -| 📋**Agenttikortin löytäminen** | `/.well-known/agent.json` asiakkaan automaattiseen löytämiseen | -| 🧪**Protokollan E2E-testivaljaat** | Todellinen MCP SDK + A2A-asiakas kulkee muodossa "test:protocols:e2e" | -| ⚙️**Toimintaohjaimet** | Vaihda yhdistelmä, käytä kimmoisuusprofiileja, nollaa katkaisijat yhdeltä ohjauspinnalta | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Ominaisuus | Mitä se tekee | -| ------------------------------------ | --------------------------------------------------------------------------------------- | ----------------------- | -| 🎯**Smart 4-Tier Fallback** | Automaattinen reitti: Tilaus → API-avain → Halpa → Ilmainen | -| 📊**Reaaliaikainen kiintiöseuranta** | Live-tunnusten määrä + nollaa lähtölaskenta palveluntarjoajaa kohti | -| 🔄**Käännösmuoto** | OpenAI ↔ Claude ↔ Gemini ↔ Vastaukset skeematurvallisilla muunnoksilla | -| 👥**Useiden tilien tuki** | Useita tilejä per palveluntarjoaja älykkäällä valinnalla | -| 🔄**Automaattinen Token Refresh** | OAuth-tunnukset päivittyvät automaattisesti yrittämällä uudelleen | -| 🎨**Muokatut yhdistelmät** | 9 tasapainotusstrategiaa + varaketjun ohjaus | -| 🌐**Wildcard-reititin** | `provider/*` dynaaminen reititys | -| 🧠**Ajatteleva budjettihallinta** | Läpivienti-, automaatti-, mukautetut ja mukautuvat päättelyrajat | -| 🔀**Mallialiakset** | Sisäänrakennettu + mukautetun mallin alias ja siirtoturva | -| ⚡**Taustan heikkeneminen** | Ohjaa matalan prioriteetin taustatehtävät halvempiin malleihin | -| 🧪**Task-Aware Smart Routing** | Automaattinen mallin valinta sisältötyypin mukaan (koodaus/näkemys/analyysi/yhteenveto) | -| 🔄**A2A-agenttityönkulut** | Deterministinen FSM-organisaattori tilallisiin monivaiheisiin agenttien suorituksiin | -| 🔀**Adaptiivinen reititys** | Dynaaminen strategian ohitus tunnuksen määrän ja nopean monimutkaisuuden perusteella | -| 🎲**Tarjoajien monimuotoisuus** | Shannonin entropiapisteytys tasapainottava automaattinen yhdistelmäliikenteen jakelu | -| 💬**Järjestelmän pikaruiskutus** | Globaalia käyttäytymisen valvontaa sovelletaan johdonmukaisesti | -| 📄**Responses API -yhteensopivuus** | Täysi "/v1/responses" tuki Codexille ja edistyneille agenttityönkuluille | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Ominaisuus | Mitä se tekee | -| ------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**Kuvan luominen** | `/v1/images/generations' pilvipalveluilla ja paikallisilla taustaohjelmilla | -| 📐**Upotukset** | `/v1/embeddings' haku- ja RAG-putkistoja varten | -| 🎤**Äänitranskriptio** | `/v1/audio/transcriptions' — 7 palveluntarjoajaa (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), automaattinen kielentunnistus, MP4/MP3/WAV-tuki | -| 🔊**Tekstistä puheeksi** | `/v1/audio/speech` – 10 palveluntarjoajaa (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) oikeilla virheilmoituksilla | -| 🎬**Videon sukupolvi** | `/v1/videos/generations' (ComfyUI + SD WebUI -työnkulut) | -| 🎵**Musiikin sukupolvi** | `/v1/music/generations' (ComfyUI-työnkulut) | -| 🛡️**Moderaatiot** | "/v1/moderations" turvallisuustarkastukset | -| 🔀**Uudelleenjärjestys** | `/v1/uudelleensijoitus' osuvuuden arvioimiseksi | -| 🔍**Verkkohaku**🆕 | "/v1/search" – 5 palveluntarjoajaa (Serper, Brave, Perplexity, Exa, Tavily), 6 500+ ilmaista kuukaudessa, automaattinen vikasieto, välimuisti | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Ominaisuus | Mitä se tekee | -| ----------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**Katkaisijat** | Mallikohtainen laukaisu/palautus kynnysohjaimilla | -| 🎯**Endpoint-Aware mallit** | Mukautetut mallit ilmoittavat tuetut päätepisteet + API-muoto | -| 🛡️**Ukkosen vastainen lauma** | Mutex + semaforisuojat uudelleenyritys-/nopeustapahtumissa | -| 🧠**Semanttinen + allekirjoitusvälimuisti** | Kustannusten/viiveen vähentäminen kahdella välimuistikerroksella | -| ⚡**Pyydä idempotenssia** | Kaksoissuojaikkuna | -| 🔒**TLS-sormenjälkien huijaus** | Selaimen kaltainen TLS-sormenjälki —**vähentää botin havaitsemista ja tilimerkintöjä** | -| 🔏**CLI-sormenjälkien vastaavuus** | Vastaa alkuperäisten CLI-pyyntöjen allekirjoituksia —**vähentää eston riskiä säilyttäen samalla välityspalvelimen IP-osoitteen** | -| 🌐**IP-suodatus** | Salli-/estoluettelon hallinta paljaille käyttöönotuksille | -| 📊**Muokattavat hintarajat** | Muokattavat globaalit/toimittajatason rajoitukset pysyvillä | -| 📉**Graceful Degradation** | Monikerroksiset valmiudet, jotka suojaavat ydinyhdyskäytävätoimintoja | -| 📜**Config Audit Trail** | Diff-pohjainen muutosseuranta, joka estää toiminnan ajautumisen yksinkertaisilla palautuksilla | -| ⏳**Provider Health Sync** | Ennakoiva tunnuksen vanhenemisen valvonta laukaisee hälytyksiä ennen valtuutusvirheitä | -| 🚪**Poista kielletyt tilit automaattisesti käytöstä** | Toiminnassa oleva katkaisija sinetöi pysyvästi estettyjen tokentilien automaattisesti | -| 🔑**API-avainten hallinta + rajaus** | Suojattu avainten myöntäminen/kierto ja mallin/toimittajan hallintalaitteet | -| 👁️**Scoped API Key Reveal**🆕 | Ota käyttöön API-avainten palautus `ALLOW_API_KEY_REVEAL` | -| 🛡️**Suojattu `/mallit`** | Valinnainen todennus ja palveluntarjoajan piilottaminen malliluetteloon | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Ominaisuus | Mitä se tekee | -| ------------------------------------------ | ----------------------------------------------------------------------- | ---------------------------- | -| 📝**Pyyntö + välityspalvelimen kirjaus** | Täysi pyyntö/vastaus ja välityspalvelimen kirjaus | -| 📉**Striimatut yksityiskohtaiset lokit**🆕 | Rekonstruoi SSE-hyötykuormavirrat puhtaasti käyttöliittymään | -| 📋**Unified Logs Dashboard** | Pyyntö-, välityspalvelin-, tarkastus- ja konsolinäkymät yhdellä sivulla | -| 🔍**Pyydä telemetriaa** | p50/p95/p99 latenssi ja pyynnön jäljitys | -| 🏥**Terveyden hallintapaneeli** | Käyttöaika, katkaisutilat, lukitukset, välimuistitilastot | -| 💰**Kustannusseuranta** | Budjetin hallinta ja mallikohtainen hinnoittelun näkyvyys | -| 📈**Analytiikan visualisoinnit** | Mallin/palveluntarjoajan käyttötiedot ja trendinäkymät | -| 🧪**Arviointikehys** | Golden set -testaus konfiguroitavilla ottelustrategioilla | -| 📡**Live Diagnostics**🆕 | Semanttisen välimuistin ohitus tarkkaan yhdistelmätestaukseen livenä | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Ominaisuus | Mitä se tekee | -| ---------------------------------- | ----------------------------------------------------------------------------------------------- | --------------------- | -| 🌐**Ota käyttöön missä tahansa** | Localhost, VPS, Docker, pilviympäristöt | -| 🚇**Cloudflare-tunneli**🆕 | Yhden napsautuksen Quick Tunnel -integrointi kojelaudalta | -| 🔑**API-avainmallin suodatus** | Alkuperäinen /v1/models-vastaus suodatettu määritettyjen verkkopalvelukontekstin roolien kautta | -| ⚡**Smart Cache Bypass** | Konfiguroitava TTL-heuristiikka ja pakotetut palautusohjaimet | -| 🔄**Varmuuskopioi/Palauta** | Vienti/tuonti ja katastrofien palautusvirrat | -| 🧙**Ohjattu käyttöönottotoiminto** | Ensimmäisen kerran ohjattu asennus | -| 🔧**CLI Tools Dashboard** | Asennus yhdellä napsautuksella suosittuja koodaustyökaluja varten | -| 🎮**Model Playground** | Testaa mitä tahansa palveluntarjoajaa/mallia/päätepistettä hallintapaneelista | -| 🔏**CLI-sormenjälkivalitsin** | Palveluntarjoajakohtainen sormenjälkien vastaavuus kohdassa Asetukset > Suojaus | -| 🌐**i18n (30 kieltä)** | Täysi kojelauta + asiakirjojen kielen tuki RTL-kattauksella | -| 🧹**Tyhjennä kaikki mallit** | Yhden napsautuksen malliluettelon tyhjennys toimittajan tiedoissa | -| 👁️**Sivupalkin säätimet**🆕 | Piilota komponentit ja integraatiot ulkoasuasetuksista | -| 📋**Ongelman mallit** | Standardoidut GitHub-mallit bugeille ja ominaisuuksille | -| 📂**Muokattu tietohakemisto** | Tallennuspaikan DATA_DIR-ohitus | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1292,103 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -Kun kiintiö, korko tai kunto epäonnistuu, OmniRoute siirtyy automaattisesti seuraavaan ehdokkaaseen ilman manuaalista vaihtoa.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A ovat löydettävissä käyttöliittymässä ja asiakirjoissa (ei piilotettu) -- Protokollan tilasovellusliittymät paljastavat reaaliaikaiset toimintatiedot (`/api/mcp/*`, `/api/a2a/*`) -- Hallintapaneelit sisältävät toimintoja 2. päivän toimintoihin (kombinvaihto, katkaisijan nollaukset, tehtävien peruutus)#### Translator + validation workflow +#### Protocol management that is visible and operable -Kääntäjä-alue sisältää: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Playground**: pyydä muunnostarkistuksia -**Chat Tester**: täydellinen pyyntö/vastaus edestakaisin -**Testipenkki**: useita tapauksia yhdellä kertaa -**Live Monitor**: reaaliaikainen liikennenäkymä +#### Translator + validation workflow -Plus protokollan validointi oikeiden asiakkaiden kanssa komennolla "npm run test:protocols:e2e". +The Translator area includes: -> 📖**[MCP-palvelimen README](open-sse/mcp-server/README.md)**— työkaluviittaus, IDE-määritykset ja asiakasesimerkit +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A-palvelimen README](src/lib/a2a/README.md)**— Taidot, JSON-RPC-menetelmät, suoratoisto ja tehtävien elinkaari## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute sisältää sisäänrakennetun arviointikehyksen, jolla testataan LLM-vastauksen laatua kultaiseen joukkoon verrattuna. Käytä sitä kojelaudan kohdassa**Analytics → Evals**.### Built-in Golden Set +## 🧪 Evaluations (Evals) -Esiladattu "OmniRoute Golden Set" sisältää testitapauksia: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Tervehdys, matematiikka, maantiede, koodin luominen -- JSON-muodon yhteensopivuus, käännös, hinnanalennusten luominen -- Turvallisuuskielto (haitallinen sisältö), laskenta, boolen logiikka### Evaluation Strategies +### Built-in Golden Set -| Strategia | Kuvaus | Esimerkki | -| ---------- | ------------------------------------------------------------------------ | ------------------------------- | --- | -| "tarkka" | Tulosten on vastattava tarkasti | "4" | -| "sisältää" | Tulosteen tulee sisältää alimerkkijono (kirjainkoolla ei ole merkitystä) | `"Pariisi"` | -| "regulex" | Tulostuksen on vastattava regex-mallia | "1.*2.*3" | -| "muokattu" | Mukautettu JS-funktio palauttaa true/false | `(lähtö) => output.length > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) - -🧩 MCP-asetukset (mallikontekstiprotokolla) +
+🧩 MCP Setup (Model Context Protocol) -Aloita MCP-siirto stdio-tilassa:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Suositeltu vahvistuskulku: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Yhdistä MCP-asiakas stdion kautta. -2. Suorita "omniroute_get_health". -3. Suorita "omniroute_list_combos". -4. Avaa `/dashboard/mcp' vahvistaaksesi syke, toiminta ja tarkastus. +Useful APIs for automation: -Hyödyllisiä sovellusliittymiä automatisointiin: +- `GET /api/mcp/status` +- `GET /api/mcp/tools` +- `GET /api/mcp/audit` +- `GET /api/mcp/audit/stats` -- "GET /api/mcp/status". -- "GET /api/mcp/tools". -- "GET /api/mcp/audit". -- "GET /api/mcp/audit/stats".
+ - -🤝 A2A-asetukset (Agent2Agent) +
+🤝 A2A Setup (Agent2Agent) -Tutustu agenttiin:```bash +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Lähetä tehtävä:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` +Manage lifecycle: -Hallitse elinkaarta: +- `GET /api/a2a/status` +- `GET /api/a2a/tasks` +- `GET /api/a2a/tasks/:id` +- `POST /api/a2a/tasks/:id/cancel` -- "GET /api/a2a/status". -- "GET /api/a2a/tasks". -- "GET /api/a2a/tasks/:id". -- POST /api/a2a/tasks/:id/cancel +Operational UI: -Käyttöliittymä: +- `/dashboard/a2a` for task/state/stream observability and smoke actions -- `/dashboard/a2a` tehtävän/tilan/virran havainnointia ja savutoimintoja varten
+ - -🧪 Päästä päähän -protokollan validointi +
+🧪 End-to-end protocol validation -Vahvista molemmat protokollat oikeilla asiakkailla:```bash +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Tämä varmistaa: +This verifies: -- MCP SDK -asiakas yhdistä/luettelo/soita -- A2A Discovery/Send/stream/get/cancel -- Tarkista tiedot MCP-tarkastuksessa ja A2A-tehtävienhallinnan sovellusliittymissä
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs - -💳 Tilauspalveluntarjoajat### Claude Code (Pro/Max) + + +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1401,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Provinkki:**Käytä Opusta monimutkaisiin tehtäviin ja Sonnetia nopeutta varten. OmniRoute jäljityskiintiö mallia kohti!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1415,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Jokaisella Codex-tilillä on nyt käytäntökytkimet kohdassa "Dashboard -> Providers": +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- "5h" (ON/OFF): pakottaa 5 tunnin ikkunan kynnyskäytäntö. -- "Viikoittain" (ON/OFF): pakottaa viikoittaisen ikkunan kynnyskäytäntöä. -- Kynnyskäyttäytyminen: kun käytössä oleva ikkuna saavuttaa >=90 % käytön, tili ohitetaan. -- Kiertokäyttäytyminen: OmniRoute reitittää automaattisesti seuraavalle kelvolliselle Codex-tilille. -- Nollauskäyttäytyminen: kun palveluntarjoajan "resetAt" aika kuluu, tili tulee uudelleen kelpoiseksi automaattisesti. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Skenaariot: +Scenarios: -- "5h PÄÄLLÄ" + "Viikoittain PÄÄLLÄ": tili ohitetaan, kun jompikumpi ikkuna saavuttaa kynnyksen. -- "5h OFF" + "Weekly ON": vain viikoittainen käyttö voi estää tilin. -- "5h ON" + "Weekly OFF": vain 5 tunnin käyttö voi estää tilin. -- `resetAt` hyväksytty: tili siirtyy uudelleen kiertoon automaattisesti (ei manuaalista uudelleenkäyttöä).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1440,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Paras hinta-laatusuhde:**Valtava ilmainen taso! Käytä tätä ennen maksettuja tasoja.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1455,71 +1662,91 @@ Models:
- -🔑 API-avaintoimittajat### NVIDIA NIM (FREE developer access — 70+ models) +
+🔑 API Key Providers -1. Rekisteröidy: [build.nvidia.com](https://build.nvidia.com) -2. Hanki ilmainen API-avain (sisältää 1000 johtopäätöskrediittiä) -3. Kojelauta → Lisää toimittaja → NVIDIA NIM: - - API-avain: "nvapi-your-key". +### NVIDIA NIM (FREE developer access — 70+ models) -**Mallit:**"nvidia/llama-3.3-70b-instruct", "nvidia/mistral-7b-instruct" ja yli 50 muuta +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Provinkki:**OpenAI-yhteensopiva API – toimii saumattomasti OmniRouten muotokäännöksen kanssa!### DeepSeek +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -1. Rekisteröidy: [platform.deepseek.com](https://platform.deepseek.com) -2. Hanki API-avain +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! + +### DeepSeek + +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key 3. Dashboard → Add Provider → DeepSeek -**Mallit:**"deepseek/deepseek-chat", "deepseek/deepseek-coder"### Groq (Free Tier Available!) +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -1. Rekisteröidy: [console.groq.com](https://console.groq.com) -2. Hanki API-avain (ilmainen taso mukana) +### Groq (Free Tier Available!) + +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) 3. Dashboard → Add Provider → Groq -**Mallit:**"groq/llama-3.3-70b", "groq/mixtral-8x7b" +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Provinkki:**Äärimmäisen nopea johtopäätös – paras reaaliaikaiseen koodaukseen!### OpenRouter (100+ Models) +**Pro Tip:** Ultra-fast inference — best for real-time coding! -1. Rekisteröidy: [openrouter.ai](https://openrouter.ai) -2. Hanki API-avain +### OpenRouter (100+ Models) + +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key 3. Dashboard → Add Provider → OpenRouter -**Mallit:**Käytä yli 100 mallia kaikilta tärkeimmiltä palveluntarjoajilta yhdellä API-avaimella. +**Models:** Access 100+ models from all major providers through a single API key. -**Kojelaudan toiminta:**OpenRouter-malleja hallitaan**Saatavilla olevista malleista**. Manuaalinen lisääminen, tuonti ja automaattinen synkronointi päivittävät kaikki saman luettelon.
+**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. - -💰 Halvat palveluntarjoajat (Varmuuskopio)### GLM-4.7 (Daily reset, $0.6/1M) + -1. Rekisteröidy: [Zhipu AI](https://open.bigmodel.cn/) -2. Hanki API-avain Coding Planista -3. Hallintapaneeli → Lisää API-avain: - - Palveluntarjoaja: "glm". - - API-avain: "oma-avain". +
+💰 Cheap Providers (Backup) -**Käytä:**`glm/glm-4.7` +### GLM-4.7 (Daily reset, $0.6/1M) -**Provinkki:**Koodaussuunnitelma tarjoaa 3-kertaisen kiintiön 1/7 hinnalla! Nollaa päivittäin klo 10.00.### MiniMax M2.1 (5h reset, $0.20/1M) +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -1. Rekisteröidy: [MiniMax](https://www.minimax.io/) -2. Hanki API-avain -3. Kojelauta → Lisää API-avain +**Use:** `glm/glm-4.7` -**Käytä:**`minimax/MiniMax-M2.1` +**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. -**Ammattilaisen vinkki:**Halvin vaihtoehto pitkälle kontekstille (1 milj. merkkiä)!### Kimi K2 ($9/month flat) +### MiniMax M2.1 (5h reset, $0.20/1M) -1. Tilaa: [Moonshot AI](https://platform.moonshot.ai/) -2. Hanki API-avain -3. Kojelauta → Lisää API-avain +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key -**Käytä:**`kimi/kimi-latest` +**Use:** `minimax/MiniMax-M2.1` -**Ammattilaisen vinkki:**Kiinteä 9 dollaria kuukaudessa 10 miljoonalle tokenille = 0,90 dollaria / 1 miljoona todellista hintaa!
+**Pro Tip:** Cheapest option for long context (1M tokens)! - -🆓 ILMAISIA palveluntarjoajia (hätävarmuuskopiointi)### Qoder (5 FREE models via OAuth) +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1560,8 +1787,10 @@ Models:
- -🎨 Luo komboja### Example 1: Maximize Subscription → Cheap Backup +
+🎨 Create Combos + +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1589,8 +1818,10 @@ Cost: $0 forever!
- -🔧 CLI-integrointi### Cursor IDE +
+🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1601,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Käytä kojelaudan**CLI Tools**-sivua määritysten tekemiseen yhdellä napsautuksella tai muokkaa `~/.claude/settings.json` manuaalisesti.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1612,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Vaihtoehto 1 – hallintapaneeli (suositus):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Vaihtoehto 2 – Manuaalinen:**Muokkaa `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1629,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Huomaa:**OpenClaw toimii vain paikallisen OmniRouten kanssa. Käytä "127.0.0.1" "localhost" sijaan IPv6-resoluutioongelmien välttämiseksi.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1643,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Vaihe 1:**Lisää OmniRoute mukautetuksi palveluntarjoajaksi:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Vaihe 2:**Luo/muokkaa `opencode.json` projektisi juuressa:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1669,117 +1909,130 @@ opencode } } } -```` +``` -**Vaihe 3:**Valitse malli OpenCodessa:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Vinkki:**Lisää mikä tahansa malli, joka on saatavilla OmniRoute `/v1/models' -päätepisteessäsi "mallit"-osioon. Käytä muotoa "provider/model-id" OmniRoute-hallintapaneelista.
+ --- ## Vianmääritys - -Laajenna vianetsintäopas napsauttamalla +
+Click to expand troubleshooting guide -**"Kielimalli ei antanut viestejä"** +**"Language model did not provide messages"** -- Palveluntarjoajan kiintiö käytetty loppuun → Tarkista kojelaudan kiintiön seuranta -- Ratkaisu: Käytä yhdistelmävaraa tai vaihda halvempaan tasoon +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**hintarajoitus** +**Rate limiting** -- Tilauskiintiö loppu → Varaa GLM/MiniMaxiin -- Lisää yhdistelmä: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**OAuth-tunnus vanhentunut** +**OAuth token expired** -- OmniRoute päivittää automaattisesti -- Jos ongelmat jatkuvat: Kojelauta → Palveluntarjoaja → Yhdistä uudelleen +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Korkeat kustannukset** +**High costs** -- Tarkista käyttötilastot kohdassa Dashboard → Costs -- Vaihda ensisijaiseksi malliksi GLM/MiniMax -- Käytä ilmaista tasoa (Gemini CLI, Qoder) ei-kriittisiin tehtäviin +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Kojelauta/API-portit ovat väärin** +**Dashboard/API ports are wrong** -- "PORT" on kanoninen perusportti (ja oletuksena API-portti) -- "API_PORT" ohittaa vain OpenAI-yhteensopivan API-kuuntelijan -- `DASHBOARD_PORT' ohittaa vain kojelaudan/Next.js-kuuntelijan -- Aseta "NEXT_PUBLIC_BASE_URL" kojelaudaksi/julkiseksi URL-osoitteeksi (OAuth-takaisinsoittoja varten) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Pilvisynkronointivirheet** +**Cloud sync errors** -- Varmista, että BASE_URL osoittaa käynnissä olevaan esiintymääsi -- Varmista, että CLOUD_URL osoittaa odotettuun pilvipäätepisteeseen -- Pidä NEXT_PUBLIC_*-arvot kohdakkain palvelinpuolen arvojen kanssa +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Ensimmäinen kirjautuminen ei toimi** +**First login not working** -- Tarkista .env:stä ALKUPERÄINEN_SALASANA -- Jos ei ole asetettu, varasalasana on "123456". +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Ei pyyntölokeja** +**No request logs** -- Pyynnön artefaktit kirjoitetaan hakemistoon DATA_DIR/call_logs/ yhtenä JSON-tiedostona pyyntöä kohden. -- Ota käyttöön liukuhihnan sieppaus kohdasta Dashboard → Logs → Request Logs, jos tarvitset yksityiskohtaisia vaihekohtaisia hyötykuormia -- Aseta APP_LOG_TO_FILE=true, jos haluat myös sovelluskonsolin lokit hakemistoon `logs/application/app.log' -- Säädä APP_LOG_MAX_FILE_SIZE, APP_LOG_RETENTION_DAYS, APP_LOG_MAX_FILES ja CALL_LOG_MAX_ENTRIES tarpeen mukaan +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Yhteystesti näyttää "Virheellinen" OpenAI-yhteensopiville palveluntarjoajille** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Monet palveluntarjoajat eivät paljasta /mallit-päätepistettä -- OmniRoute v1.0.6+ sisältää varatarkistuksen chatin loppuunsaattamisen kautta -- Varmista, että perus-URL sisältää /v1-liitteen### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ Tärkeää käyttäjille, jotka käyttävät OmniRoutea VPS:ssä, Dockerissa tai millä tahansa etäpalvelimella**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -**Antigravity**- ja**Gemini CLI**-palveluntarjoajat käyttävät**Google OAuth 2.0**-versiota. Google edellyttää, että OAuth-kulun "redirect_uri" vastaa täsmälleen yhtä sovelluksen Google Cloud Consolessa esirekisteröityistä URI:ista. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -OmniRouteen niputetut OAuth-tunnistetiedot on rekisteröity**vain `localhost'ille**. Kun käytät OmniRoutea etäpalvelimella (esim. `https://omniroute.myserver.com`), Google hylkää todennuksen seuraavilla tavoilla:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Sinun on luotava**OAuth 2.0 -asiakastunnus**Google Cloud Consolessa palvelimesi URI:lla.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Avaa Google Cloud Console** +#### Step-by-step -Siirry osoitteeseen: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Luo uusi OAuth 2.0 -asiakastunnus** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Napsauta**"+ Luo kirjautumistiedot"**→**"OAuth-asiakastunnus"** -- Sovellustyyppi:**"Web-sovellus"** -- Nimi: kaikki mistä pidät (esim. "OmniRoute Remote") +**2. Create a new OAuth 2.0 Client ID** + +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) **3. Add Authorized Redirect URIs** -Lisää**"Authorized redirect URIs"**-kenttään:``` +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Korvaa "your-server.com" palvelimesi verkkotunnuksella tai IP-osoitteella (lisää tarvittaessa portti, esim. "http://45.33.32.156:20128/callback"). +**4. Save and copy the credentials** -**4. Tallenna ja kopioi tunnistetiedot** +After creating, Google will show the **Client ID** and **Client Secret**. -Luomisen jälkeen Google näyttää**Client ID**ja**Client Secret**. +**5. Set environment variables** -**5. Aseta ympäristömuuttujat** +In your `.env` (or Docker environment variables): -.env-tiedostossa (tai Docker-ympäristömuuttujat):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1788,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Käynnistä OmniRoute uudelleen**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Yritä muodostaa yhteys uudelleen** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Dashboard → Providers → Antigravity (tai Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google ohjaa nyt oikein osoitteeseen https://your-server.com/callback.--- +--- #### Temporary workaround (without custom credentials) -Jos et halua määrittää omia tunnistetietojasi juuri nyt, voit silti käyttää**manuaalista URL-kulkua**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute avaa Googlen valtuutus-URL-osoitteen -2. Valtuutuksen jälkeen Google yrittää uudelleenohjata palvelimeen "localhost" (joka epäonnistuu etäpalvelimella) -3.**Kopioi koko URL-osoite**selaimesi osoitepalkista (vaikka sivu ei latautuisi) -4. Liitä URL-osoite OmniRoute-yhteysmodaalissa näkyvään kenttään -5. Napsauta**"Yhdistä"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Tämä toimii, koska URL-osoitteessa oleva valtuutuskoodi on kelvollinen riippumatta siitä, ladattiinko uudelleenohjaussivu.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. - -🇧🇷 Versão em Português#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Os provedores**Antigravity**ja**Gemini CLI**usam**Google OAuth 2.0**para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja**exatamente**uma das URIs pre-cadastradas no Google Cloud Console do aplicativo. +
+🇧🇷 Versão em Português -As credenciais OAuth embutidas no OmniRoute estão cadastradas**apenas para `localhost`**. Quando você acessa tai OmniRoute em um servidor Remoto (esim. `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Você precisa criar um**OAuth 2.0 Client ID**ei Google Cloud Console com URI do seu servidor.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Acesse tai Google Cloud Console** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -**2. Crie um novo OAuth 2.0 -asiakastunnus** +**2. Crie um novo OAuth 2.0 Client ID** -- Klikkaa em**"+ Luo kirjautumistiedot"**→**"OAuth-asiakastunnus"** -- Tipo de aplicativo:**"Web-sovellus"** -- Nimi: escolha qualquer nome (esim. "OmniRoute Remote") +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Adicione valtuutettuina uudelleenohjaus-URI:ina** +**3. Adicione as Authorized Redirect URIs** -No campo**"Authorized redirect URIs"**, lisäys:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Korvaa `seu-servidor.com` pelo domínio tai IP do seu servidor (mukaan lukien porta se necessário, esim. `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. Tallenna kopio valtuutuksena** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Após criar, o Google mostrará o**Client ID**e o**Client Secret**. +**5. Configure as variáveis de ambiente** -**5. Määritä variáveis de ambiente** +No seu `.env` (ou nas variáveis de ambiente do Docker): -Ei seu `.env` (ou nas variáveis de ambiente do Docker):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1867,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reinicie o OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute - -```` +``` **7. Tente conectar novamente** Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará.--- +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. + +--- #### Workaround temporário (sem configurar credenciais próprias) -Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo**manual de URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. OmniRoute lähettää Googlen lupa-osoitteen -2. Após você autorizar, o Google tentará redirecionar para "localhost" (que falha no servidor Remoto) -3.**Kopioi URL-osoite täydellinen**da barra de endereço do seu selaimessa (mesmo que a página não carregue) +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) 4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute -5. Klikkaa em**"Yhdistä"** +5. Clique em **"Connect"** -> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1905,64 +2171,73 @@ Se não quiser criar credenciais próprias agora, ainda é possível usar o flux ## 🛠️ Tech Stack - +
Click to expand tech stack details --**Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is**not supported**— `better-sqlite3` native binaries are incompatible) --**Language**: TypeScript 5.9 —**100% TypeScript**across `src/` and `open-sse/` (zero `any` in core modules since v2.0) --**Framework**: Next.js 16 + React 19 + Tailwind CSS 4 --**Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) --**Schemas**: Zod (MCP tool I/O validation, API contracts) --**Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Striimaus**: Palvelimen lähettämät tapahtumat (SSE) --**Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization --**Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) --**CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) --**Verkkosivusto**: [omniroute.online](https://omniroute.online) --**Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Dokumentaatio -| Asiakirja | Kuvaus | -| ----------------------------------------------- | ---------------------------------------------------- | -| [Käyttöopas](docs/USER_GUIDE.md) | Palveluntarjoajat, yhdistelmät, CLI-integrointi, käyttöönotto | -| [API-viite](docs/API_REFERENCE.md) | Kaikki päätepisteet esimerkeineen | -| [MCP-palvelin](open-sse/mcp-server/README.md) | 16 MCP-työkalua, IDE-konfiguraatioita, Python/TS/Go-asiakkaita | -| [A2A-palvelin](src/lib/a2a/README.md) | JSON-RPC 2.0 -protokolla, taidot, suoratoisto, tehtävänhallinta | -| [Auto-Combo Engine](docs/auto-combo.md) | 6-faktorinen pisteytys, tilapaketit, itsestään paraneva | -| [Vianetsintä](docs/TROUBLESHOOTING.md) | Yleisiä ongelmia ja ratkaisuja | -| [Arkkitehtuuri](docs/ARCHITECTURE.md) | Järjestelmäarkkitehtuuri ja sisäosat | -| [Osallistuva](CONTRIBUTING.md) | Kehittämisjärjestelyt ja -ohjeet | -| [OpenAPI-määritys](docs/openapi.yaml) | OpenAPI 3.0 -spesifikaatio | -| [Turvallisuuspolitiikka](SECURITY.md) | Haavoittuvuusraportointi ja tietoturvakäytännöt | -| [VM-käyttöönotto](docs/VM_DEPLOYMENT_GUIDE.md) | Täydellinen opas: VM + nginx + Cloudflare-asennus | -| [Ominaisuudet Galleria](docs/FEATURES.md) | Visuaalinen kojelautakierros kuvakaappauksilla | -| [Julkaisun tarkistuslista](docs/RELEASE_CHECKLIST.md) | Julkaisua edeltävän vahvistuksen vaiheet |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoutella on**210+ suunniteltua ominaisuutta**useissa kehitysvaiheissa. Tässä ovat tärkeimmät alueet: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Category | Suunnitellut ominaisuudet | Kohokohdat | -| ------------------------------ | ----------------- | --------------------------------------------------------------------------------------- | -| 🧠**Routing & Intelligence**| 25+ | Pienimmän viiveen reititys, tunnistepohjainen reititys, kiintiön esitarkastus, P2C-tilin valinta | -| 🔒**Turvallisuus ja vaatimustenmukaisuus**| 20+ | SSRF-karkaisu, valtuustietojen peittäminen, päätepistekohtainen nopeusraja, hallintaavaimen laajuus | -| 📊**Havaittavuus**| 15+ | OpenTelemetry-integraatio, reaaliaikainen kiintiöiden seuranta, kustannusseuranta mallikohtaisesti | -| 🔄**Tarjoajien integraatiot**| 20+ | Dynaaminen mallirekisteri, palveluntarjoajan jäähtyminen, usean tilin Codex, Copilot-kiintiön jäsentäminen | -| ⚡**Suorituskyky**| 15+ | Kaksoisvälimuistikerros, kehotevälimuisti, vastausvälimuisti, suoratoiston ylläpitäminen, erä-API | -| 🌐**Ekosysteemi**| 10+ | WebSocket API, konfiguroinnin hot-reload, hajautettu konfiguraatiosäilö, kaupallinen tila |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**OpenCode Integration**- Natiivitoimittajan tuki OpenCode AI -koodaus-IDE:lle -- 🔗**TRAE-integraatio**— Täysi tuki TRAE AI -kehityskehykselle -- 📦**Eräsovellusliittymä**— Asynkroninen eräkäsittely joukkopyyntöille -- 🎯**Tagipohjainen reititys**- Reittipyynnöt mukautettujen tunnisteiden ja metatietojen perusteella -- 💰**Alhaisimman kustannustason strategia**- Valitse automaattisesti halvin saatavilla oleva palveluntarjoaja +### 🔜 Coming Soon -> 📝 Täydelliset ominaisuudet saatavilla osoitteesta [`docs/new-features/`](docs/new-features/) (217 yksityiskohtaista spesifikaatiota)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1970,18 +2245,20 @@ OmniRoutella on**210+ suunniteltua ominaisuutta**useissa kehitysvaiheissa. Täss ### How to Contribute -1. Haarukka arkisto -2. Luo ominaisuushaara (`git checkout -b feature/amazing-feature`) -3. Vahvista muutokset (`git commit -m 'Lisää upea ominaisuus') -4. Työnnä haaraan (`git push origin ominaisuus/amazing-feature`) -5. Avaa vetopyyntö +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Katso tarkemmat ohjeet kohdasta [CONTRIBUTING.md](CONTRIBUTING.md).### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1993,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Erityinen kiitos**[decolua](https://github.com/decolua)\*\***[9router](https://github.com/decolua/9router)\*\*– alkuperäiselle projektille, joka inspiroi tätä haarukkaa. OmniRoute rakentaa tälle uskomattomalle perustalle lisäominaisuuksia, multimodaalisia sovellusliittymiä ja täydellistä TypeScript-uudelleenkirjoitusta. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Erityiset kiitokset**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**-sovellukselle, joka on alkuperäinen Go-toteutus, joka inspiroi tätä JavaScript-porttia.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Lisenssi -MIT-lisenssi – katso [LICENSE](LICENSE) saadaksesi lisätietoja.--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/fi/docs/ARCHITECTURE.md b/docs/i18n/fi/docs/ARCHITECTURE.md index c0d7433ec1..2b20d116ee 100644 --- a/docs/i18n/fi/docs/ARCHITECTURE.md +++ b/docs/i18n/fi/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Viimeksi päivitetty: 2026-03-28_## Executive Summary -OmniRoute on paikallinen AI-reititysyhdyskäytävä ja kojelauta, joka on rakennettu Next.js:lle. -Se tarjoaa yhden OpenAI-yhteensopivan päätepisteen (`/v1/*`) ja reitittää liikenteen useiden alkupään palveluntarjoajien kesken kääntämisen, varaajan, tunnuksen päivityksen ja käytön seurannan avulla. -Ydinominaisuudet: +_Last updated: 2026-03-28_ -- OpenAI-yhteensopiva API-pinta CLI:lle/työkaluille (28 toimittajaa) -- Pyydä/vastaa käännös palveluntarjoajan eri formaattien välillä -- Mallin yhdistelmävara (usean mallin sarja) -- Tilitason varatoiminto (usea tili palveluntarjoajaa kohti) -- OAuth + API-avain tarjoajan yhteyden hallinta -- Upottamisen luominen /v1/embeddings-tiedoston kautta (6 palveluntarjoajaa, 9 mallia) -- Kuvien luominen /v1/images/generations-tiedoston kautta (4 toimittajaa, 9 mallia) -- Ajattele tagien jäsentämistä (`...`) päättelymalleille -- Vastauksen desinfiointi tiukan OpenAI SDK -yhteensopivuuden takaamiseksi -- Roolien normalisointi (kehittäjä→järjestelmä, järjestelmä→käyttäjä) palveluntarjoajien välistä yhteensopivuutta varten -- Strukturoitu lähdön muunnos (json_schema → Gemini responseSchema) -- Paikallinen pysyvyys tarjoajille, avaimille, aliaksille, yhdistelmille, asetuksille, hinnoittelulle -- Käytön/kustannusten seuranta ja pyyntöjen kirjaaminen -- Valinnainen pilvisynkronointi usean laitteen/tilan synkronointiin -- IP-sallitut / estolistat API-käyttöoikeuksien hallinnassa -- Ajatteleva budjetin hallinta (läpivienti/automaattinen/mukautettu/mukautuva) -- Globaali järjestelmän nopea ruiskutus -- Istunnon seuranta ja sormenjäljet -- Tilikohtainen tehostettu hintarajoitus tarjoajakohtaisilla profiileilla -- Katkaisijakuvio palveluntarjoajan joustavuuden parantamiseksi -- Ukkosta estävä laumasuoja mutex-lukolla -- Allekirjoituspohjainen pyyntöjen duplikoinnin välimuisti -- Verkkotunnustaso: mallin saatavuus, hintasäännöt, varakäytäntö, lukituskäytäntö -- Verkkotunnuksen tilan pysyvyys (SQLite-kirjoitusvälimuisti varauksille, budjeteille, lukituksille, katkaisimille) -- Käytäntömoottori keskitettyä pyyntöjen arviointia varten (sulku → budjetti → vara) -- Pyydä telemetriaa p50/p95/p99-latenssiaggregaatiolla -- Korrelaatiotunnus (X-Request-Id) päästä päähän -jäljitykseen -- Vaatimustenmukaisuuden tarkastuksen kirjaaminen ja opt-out API-avaimella -- Eval-kehys LLM-laadunvarmistukseen -- Joustavan käyttöliittymän kojelauta, jossa on reaaliaikainen katkaisijatila -- Modulaariset OAuth-palveluntarjoajat (12 yksittäistä moduulia kohdassa "src/lib/oauth/providers/") +## Executive Summary -Ensisijainen suoritusaikamalli: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- Next.js-sovellusreitit kohdassa `src/app/api/*` toteuttavat sekä hallintapaneelin sovellusliittymiä että yhteensopivuussovellusliittymiä -- Jaettu SSE/reititysydin kohdassa `src/sse/*` + `open-sse/*` hoitaa palveluntarjoajan suorittamisen, käännöksen, suoratoiston, varatoiminnon ja käytön## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Paikallisen yhdyskäytävän suoritusaika -- Kojelaudan hallintasovellusliittymät -- Palveluntarjoajan todennus ja tunnuksen päivitys -- Pyydä käännöstä ja SSE-suoratoistoa -- Paikallinen tila + käytön pysyvyys -- Valinnainen pilvisynkronointiorkesteri### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Pilvipalvelun toteutus osoitteen "NEXT_PUBLIC_CLOUD_URL" takana -- Palveluntarjoajan SLA/ohjaustaso paikallisen prosessin ulkopuolella -- Itse ulkoiset CLI-binaarit (Claude CLI, Codex CLI jne.)## Dashboard Surface (Current) +### Out of Scope -Pääsivut kohdassa `src/app/(dashboard)/dashboard/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — pika-aloitus + palveluntarjoajan yleiskatsaus -- "/dashboard/endpoint" - päätepisteen välityspalvelin + MCP + A2A + API-päätepisteen välilehdet -- "/dashboard/providers" — palveluntarjoajan yhteydet ja tunnistetiedot -- "/dashboard/combos" - yhdistelmästrategiat, mallit, mallin reitityssäännöt -- "/dashboard/costs" — kustannusten yhteenlaskettu ja hinnoittelun näkyvyys -- `/dashboard/analytics' — käyttöanalytiikka ja arvioinnit -- "/dashboard/limits" - kiintiön/hinnan säätimet -- "/dashboard/cli-tools" - CLI:n käyttöönotto, suorituksenaikainen tunnistus, asetusten luominen -- "/dashboard/agents" — havaitut ACP-agentit + mukautetun agentin rekisteröinti -- `/dashboard/media` — kuvan/videon/musiikin leikkipaikka -- `/dashboard/search-tools' — hakupalveluntarjoajan testaus ja historia -- `/dashboard/health' — käytettävyysaika, katkaisijat, nopeusrajoitukset -- "/dashboard/logs" — pyyntö/välityspalvelin/tarkastus/konsolilokit -- "/dashboard/settings" — järjestelmäasetusten välilehdet (yleiset, reititys, yhdistelmäoletusasetukset jne.) -- `/dashboard/api-manager` — API-avaimen elinkaaren ja mallin käyttöoikeudet## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Päähakemistot: +Main directories: -- `src/app/api/v1/*` ja `src/app/api/v1beta/*` yhteensopiville sovellusliittymille -- `src/app/api/*` hallinta-/määrityssovellusliittymille -- Seuraavaksi kirjoitetaan uudelleen `next.config.mjs`-kartassa `/v1/*` muotoon `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Tärkeitä yhteensopivuusreittejä: +Important compatibility routes: -- `src/app/api/v1/chat/completions/route.ts' -- "src/app/api/v1/messages/route.ts". -- "src/app/api/v1/responses/route.ts". -- "src/app/api/v1/models/route.ts" - sisältää mukautettuja malleja "custom: true" -- "src/app/api/v1/embeddings/route.ts" - upotuksen sukupolvi (6 tarjoajaa) -- "src/app/api/v1/images/generations/route.ts" - kuvien luominen (4+ tarjoajaa, mukaan lukien Antigravity/Nebius) +- `src/app/api/v1/chat/completions/route.ts` +- `src/app/api/v1/messages/route.ts` +- `src/app/api/v1/responses/route.ts` +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- "src/app/api/v1/providers/[provider]/chat/completions/route.ts" - palveluntarjoajakohtainen keskustelu -- `src/app/api/v1/providers/[provider]/embeddings/route.ts' – omat palveluntarjoajakohtaiset upotukset -- "src/app/api/v1/providers/[provider]/images/generations/route.ts" - palveluntarjoajakohtaiset kuvat -- "src/app/api/v1beta/models/route.ts". -- `src/app/api/v1beta/models/[...polku]/route.ts` +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images +- `src/app/api/v1beta/models/route.ts` +- `src/app/api/v1beta/models/[...path]/route.ts` -Hallintoverkkotunnukset: +Management domains: -- Todennus/asetukset: `src/app/api/auth/*`, `src/app/api/settings/*` -- Palveluntarjoajat/yhteydet: `src/app/api/providers*` -- Palveluntarjoajan solmut: `src/app/api/provider-nodes\*' -- Mukautetut mallit: `src/app/api/provider-models' (GET/POST/DELETE) -- Malliluettelo: `src/app/api/models/route.ts' (GET) -- Välityspalvelimen konfiguraatio: "src/app/api/settings/proxy" (GET/PUT/DELETE) + "src/app/api/settings/proxy/test" (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- Keys/aliases/combos/pricing: "src/app/api/keys*", "src/app/api/models/alias", "src/app/api/combos*", "src/app/api/pricing" -- Käyttö: `src/app/api/usage/*` -- Synkronointi/pilvi: `src/app/api/sync/*`, `src/app/api/cloud/*` -- CLI-työkalujen apuohjelmat: `src/app/api/cli-tools/*` -- IP-suodatin: `src/app/api/settings/ip-filter' (GET/PUT) -- Thinking-budjetti: `src/app/api/settings/thinking-budget' (GET/PUT) -- Järjestelmäkehote: `src/app/api/settings/system-prompt' (GET/PUT) -- Istunnot: `src/app/api/sessions' (GET) -- Nopeusrajoitukset: `src/app/api/rate-limits' (GET) -- Kestävyys: "src/app/api/resilience" (GET/PATCH) – palveluntarjoajan profiilit, katkaisija, nopeusrajoitustila -- Kestävyyden nollaus: "src/app/api/resilience/reset" (POST) - nollaa katkaisijat + jäähdytys -- Välimuistitilastot: `src/app/api/cache/stats' (GET/DELETE) -- Mallin saatavuus: `src/app/api/models/availability' (GET/POST) -- Telemetria: "src/app/api/telemetry/summary" (GET) -- Budjetti: `src/app/api/usage/budget' (GET/POST) -- Varaketjut: `src/app/api/fallback/chains' (GET/POST/DELETE) -- Vaatimustenmukaisuuden tarkastus: `src/app/api/compliance/audit-log' (GET) -- Evals: "src/app/api/evals" (GET/POST), "src/app/api/evals/[suiteId]" (GET) -- Käytännöt: `src/app/api/policies' (GET/POST)## 2) SSE + Translation Core +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -Päävirtausmoduulit: +## 2) SSE + Translation Core -- Merkintä: "src/sse/handlers/chat.ts". -- Ydinorkesteri: `open-sse/handlers/chatCore.ts` -- Tarjoajan suoritussovittimet: `open-sse/executors/*` -- Muototunnistuksen/palveluntarjoajan kokoonpano: `open-sse/services/provider.ts` -- Mallin jäsennys/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Tilin varalogiikka: "open-sse/services/accountFallback.ts". -- Käännösrekisteri: "open-sse/translator/index.ts". -- Stream-muunnokset: "open-sse/utils/stream.ts", "open-sse/utils/streamHandler.ts" -- Käytön purkaminen/normalisointi: `open-sse/utils/usageTracking.ts` -- Think tag -parser: `open-sse/utils/thinkTagParser.ts` -- Upotuskäsittelijä: "open-sse/handlers/embeddings.ts". -- Upotuspalveluntarjoajan rekisteri: `open-sse/config/embeddingRegistry.ts` -- Kuvanluontikäsittelijä: "open-sse/handlers/imageGeneration.ts". -- Kuvantarjoajan rekisteri: "open-sse/config/imageRegistry.ts". -- Vastauksen desinfiointi: "open-sse/handlers/responseSanitizer.ts" -- Roolin normalisointi: "open-sse/services/roleNormalizer.ts". +Main flow modules: -Palvelut (liiketoimintalogiikka): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- Tilin valinta/pisteytys: `open-sse/services/accountSelector.ts` -- Kontekstin elinkaarihallinta: `open-sse/services/contextManager.ts` -- IP-suodattimen valvonta: "open-sse/services/ipFilter.ts". -- Istunnon seuranta: `open-sse/services/sessionManager.ts` -- Pyydä päällekkäisyyden poistoa: `open-sse/services/signatureCache.ts` -- Järjestelmäkehotteen lisäys: "open-sse/services/systemPrompt.ts". -- Ajatteleva budjetin hallinta: "open-sse/services/thinkingBudget.ts" -- Jokerimerkkimallin reititys: `open-sse/services/wildcardRouter.ts` -- Hintarajan hallinta: `open-sse/services/rateLimitManager.ts` -- Katkaisija: "open-sse/services/circuitBreaker.ts" +Services (business logic): -Domain-kerroksen moduulit: +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- Mallin saatavuus: "src/lib/domain/modelAvailability.ts". -- Kustannussäännöt/budjetit: `src/lib/domain/costRules.ts` -- Varakäytäntö: "src/lib/domain/fallbackPolicy.ts". -- Yhdistelmäratkaisu: `src/lib/domain/comboResolver.ts' -- Lukituskäytäntö: "src/lib/domain/lockoutPolicy.ts". -- Käytäntömoottori: "src/domain/policyEngine.ts" — keskitetty lukitus → budjetti → varaarviointi -- Virhekoodiluettelo: "src/lib/domain/errorCodes.ts". -- Pyyntötunnus: `src/lib/domain/requestId.ts' -- Haun aikakatkaisu: "src/lib/domain/fetchTimeout.ts". -- Pyydä telemetriaa: `src/lib/domain/requestTelemetry.ts` -- Vaatimustenmukaisuus/tarkastus: `src/lib/domain/compliance/index.ts' +Domain layer modules: + +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` - Eval runner: `src/lib/domain/evalRunner.ts` -- Verkkotunnuksen tilan pysyvyys: `src/lib/db/domainState.ts' — SQLite CRUD varaketjuille, budjeteille, kustannushistorialle, lukitustilalle, katkaisimille +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -OAuth-palveluntarjoajan moduulit (12 yksittäistä tiedostoa kohdassa "src/lib/oauth/providers/"): +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -- Rekisterihakemisto: "src/lib/oauth/providers/index.ts". -- Yksittäiset palveluntarjoajat: "claude.ts", "codex.ts", "gemini.ts", "antigravity.ts", "qoder.ts", "qwen.ts", "kimi-coding.ts", "github.ts", "kiro.cursorts", `cline.ts` -- Ohut kääre: "src/lib/oauth/providers.ts" - uudelleenvienti yksittäisistä moduuleista## 3) Persistence Layer +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -Ensisijainen tila DB (SQLite): +## 3) Persistence Layer -- Ydininfrastruktuuri: "src/lib/db/core.ts" (better-sqlite3, migraatiot, WAL) -- Vie julkisivu uudelleen: "src/lib/localDb.ts" (ohut yhteensopivuuskerros soittajille) -- tiedosto: `${DATA_DIR}/storage.sqlite` (tai `$XDG_CONFIG_HOME/omniroute/storage.sqlite`, kun se on asetettu, muuten `~/.omniroute/storage.sqlite`) -- entiteetit (taulukot + KV-nimitilat): providerConnections, providerNodes, mallialiakset, yhdistelmät, apiKeys, asetukset, hinnoittelu,**customModels**,**proxyConfig**,**ipFilter**,**thhinkingBudget**,**systemPrompt** +Primary state DB (SQLite): -Käytön pysyvyys: +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -- julkisivu: "src/lib/usageDb.ts" (hajotetut moduulit tiedostossa "src/lib/usage/\*") -- SQLite-taulukot tiedostossa "storage.sqlite": "usage_history", "call_logs", "proxy_logs" -- valinnaiset tiedostoartefaktit jäävät yhteensopivuutta/virheenkorjausta varten (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- Vanhat JSON-tiedostot siirretään SQLiteen käynnistyssiirroilla, kun ne ovat olemassa +Usage persistence: + +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present Domain State DB (SQLite): -- "src/lib/db/domainState.ts" - CRUD-toiminnot toimialueen tilalle -- Taulukot (luodut tiedostossa "src/lib/db/core.ts"): "domain_fallback_chains", "domain_budgets", "domain_cost_history", "domain_lockout_state", "domain_circuit_breakers" -- Kirjoitusvälimuistin malli: muistissa olevat kartat ovat arvovaltaisia ajon aikana; mutaatiot kirjoitetaan synkronisesti SQLiten kanssa; tila palautetaan DB:stä kylmäkäynnistyksen yhteydessä## 4) Auth + Security Surfaces +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start -- Hallintapaneelin evästeiden todennus: "src/proxy.ts", "src/app/api/auth/login/route.ts" -- API-avaimen luonti/vahvistus: `src/shared/utils/apiKey.ts` -- Palveluntarjoajan salaisuudet säilyivät "providerConnections"-merkinnöissä -- Lähtevän välityspalvelimen tuki "open-sse/utils/proxyFetch.ts" (env vars) ja "open-sse/utils/networkProxy.ts" kautta (määritettävä palveluntarjoajakohtaisesti tai globaali)## 5) Cloud Sync +## 4) Auth + Security Surfaces -- Aikataulun aloitus: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Säännöllinen tehtävä: `src/shared/services/cloudSyncScheduler.ts` -- Säännöllinen tehtävä: `src/shared/services/modelSyncScheduler.ts` -- Hallitse reittiä: `src/app/api/sync/cloud/route.ts'## Request Lifecycle (`/v1/chat/completions`) +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Varapäätökset tehdään "open-sse/services/accountFallback.ts":n avulla tilakoodeja ja virheviestiheuristiikkaa käyttäen. Yhdistelmäreititys lisää yhden ylimääräisen suojan: palveluntarjoajan kattamat 400:t, kuten ylävirran sisällön lohko- ja roolivahvistuksen epäonnistumiset, käsitellään mallin paikallisina virheinä, jotta myöhempiä yhdistelmäkohteita voidaan edelleen suorittaa.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -Päivitys reaaliaikaisen liikenteen aikana suoritetaan "open-sse/handlers/chatCore.ts" -tiedostossa suorittimen "refreshCredentials()" kautta.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -"CloudSyncScheduler" käynnistää säännöllisen synkronoinnin, kun pilvi on käytössä.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -Fyysiset tallennustiedostot: +Physical storage files: -- ensisijainen ajonaikainen tietokanta: `${DATA_DIR}/storage.sqlite` -- pyyntölokin rivit: `${DATA_DIR}/log.txt` (compat/debug artefact) -- jäsennellyt puhelun hyötykuorma-arkistot: `${DATA_DIR}/call_logs/` -- valinnainen kääntäjä/pyydä virheenkorjausistuntoja: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: yhteensopivuussovellusliittymät -- `src/app/api/v1/providers/[provider]/*`: omat palveluntarjoajakohtaiset reitit (chat, upotukset, kuvat) -- `src/app/api/providers\*: palveluntarjoajan CRUD, validointi, testaus -- `src/app/api/provider-nodes\*: mukautettu yhteensopiva solmuhallinta -- "src/app/api/provider-models": mukautetun mallin hallinta (CRUD) -- "src/app/api/models/route.ts": malliluettelon sovellusliittymä (aliakset + mukautetut mallit) -- `src/app/api/oauth/*`: OAuth/laitekoodikulku -- `src/app/api/keys\*: paikallisen API-avaimen elinkaari -- "src/app/api/models/alias": aliaksen hallinta -- `src/app/api/combos*`: varayhdistelmähallinta -- "src/app/api/pricing": hinnoittelu ohittaa kustannuslaskennan -- "src/app/api/settings/proxy": välityspalvelimen määritykset (GET/PUT/DELETE) -- "src/app/api/settings/proxy/test": lähtevän välityspalvelimen yhteystesti (POST) -- `src/app/api/usage/*`: käyttö- ja lokisovellusliittymät -- `src/app/api/sync/*` + `src/app/api/cloud/*`: pilvisynkronointi ja pilveen suuntautuvat apuohjelmat -- `src/app/api/cli-tools/*`: paikalliset CLI-asetusten kirjoittajat/tarkistajat -- `src/app/api/settings/ip-filter': IP-sallittujen luettelo/estolista (GET/PUT) -- `src/app/api/settings/thhinking-budget': ajattelutunnuksen budjetin konfiguraatio (GET/PUT) -- "src/app/api/settings/system-prompt": yleinen järjestelmäkehote (GET/PUT) -- `src/app/api/sessions': aktiivisten istuntojen luettelo (GET) -- "src/app/api/rate-limits": tilikohtainen korkorajoitustila (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts: pyynnön jäsennys, yhdistelmäkäsittely, tilin valintasilmukka -- `open-sse/handlers/chatCore.ts`: käännös, suorittimen lähettäminen, uudelleenyritys/päivityskäsittely, streamin määritys -- `open-sse/executors/*`: palveluntarjoajakohtainen verkko- ja muotokäyttäytyminen### Translation Registry and Format Converters +### Routing and Execution Core -- "open-sse/translator/index.ts": kääntäjien rekisteri ja orkestrointi -- Pyydä kääntäjiä: `open-sse/translator/request/*` -- Vastauskääntäjät: `open-sse/translator/response/*` -- Muotovakiot: "open-sse/translator/formats.ts".### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: pysyvä konfiguraatio/tila ja verkkotunnuksen pysyvyys SQLitessa -- `src/lib/localDb.ts`: DB-moduulien yhteensopivuuden uudelleenvienti -- `src/lib/usageDb.ts`: käyttöhistorian/puhelulokien julkisivu SQLite-taulukoiden päällä## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Jokaisella palveluntarjoajalla on erikoistunut suorittaja, joka laajentaa "BaseExecutoria" (hakemistossa "open-sse/executors/base.ts"), joka tarjoaa URL-osoitteen rakentamisen, otsikon rakentamisen, uudelleenyrityksen eksponentiaalisella perääntymisellä, valtuustietojen päivityskoukut ja execute()-orkesterimenetelmän. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Toteuttaja | Palveluntarjoaja(t) | Erikoiskäsittely | -| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------- | -| "DefaultExecutor" | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, ilotulitus, Cerebras, Cohere, NVIDIA | Dynaaminen URL-/otsikkomääritykset tarjoajakohtaisesti | -| "AntigravityExecutor" | Google Antigravity | Mukautetut projekti-/istuntotunnukset, Yritä uudelleen jäsentämisen jälkeen | -| "CodexExecutor" | OpenAI Codex | Syöttää järjestelmäohjeita, pakottaa päättelyponnistuksen | -| "CursorExecutor" | Kohdistin IDE | ConnectRPC-protokolla, Protobuf-koodaus, pyynnön allekirjoitus tarkistussumman kautta | -| "GithubExecutor" | GitHub Copilot | Copilot-tunnuksen päivitys, VSC-koodia jäljittelevät otsikot | -| "KiroExecutor" | AWS CodeWhisperer/Kiro | AWS EventStream binaarimuoto → SSE-muunnos | -| "GeminiCLIExecutor" | Gemini CLI | Google OAuth -tunnuksen päivitysjakso | +### Persistence -Kaikki muut palveluntarjoajat (mukaan lukien mukautetut yhteensopivat solmut) käyttävät DefaultExecutoria.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Palveluntarjoaja | Muoto | Auth | Striimaa | Ei-stream | Token Refresh | Käyttösovellusliittymä | -| ---------------- | ----------------- | ------------------------- | -------------------- | --------- | ------------- | --------------------------- | ------------------------------ | -| Claude | claude | API-avain / OAuth | ✅ | ✅ | ✅ | ⚠️ Vain järjestelmänvalvoja | -| Kaksoset | kaksoset | API-avain / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | -| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | -| Antigravitaatio | antigravitaatio | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | -| OpenAI | openai | API-avain | ✅ | ✅ | ❌ | ❌ | -| Codex | openai-vastaukset | OAuth | ✅ pakotettu | ❌ | ✅ | ✅ Hintarajat | -| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Kiintiön tilannekuvat | -| Kursori | kohdistin | Mukautettu tarkistussumma | ✅ | ✅ | ❌ | ❌ | -| Kiro | kiro | AWS SSO OIDC | ✅ (TapahtumaStream) | ❌ | ✅ | ✅ Käyttörajoitukset | -| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Pyynnöstä | -| Qoder | openai | OAuth (Perus) | ✅ | ✅ | ✅ | ⚠️ Pyynnöstä | -| OpenRouter | openai | API-avain | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | claude | API-avain | ✅ | ✅ | ❌ | ❌ | -| DeepSeek | openai | API-avain | ✅ | ✅ | ❌ | ❌ | -| Groq | openai | API-avain | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | openai | API-avain | ✅ | ✅ | ❌ | ❌ | -| Mistral | openai | API-avain | ✅ | ✅ | ❌ | ❌ | -| Hämmennys | openai | API-avain | ✅ | ✅ | ❌ | ❌ | -| Yhdessä AI | openai | API-avain | ✅ | ✅ | ❌ | ❌ | -| Ilotulitus AI | openai | API-avain | ✅ | ✅ | ❌ | ❌ | -| Aivot | openai | API-avain | ✅ | ✅ | ❌ | ❌ | -| Cohere | openai | API-avain | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | openai | API-avain | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Havaittuja lähdemuotoja ovat: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. -- "openai". -- "openai-vastaukset". -- "claude". -- "kaksoset". +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | -Kohdemuotoja ovat: +All other providers (including custom compatible nodes) use the `DefaultExecutor`. -- OpenAI chat / vastaukset +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: + +- `openai` +- `openai-responses` +- `claude` +- `gemini` + +Target formats include: + +- OpenAI chat/Responses - Claude -- Gemini/Gemini-CLI/Antigravity-kuori +- Gemini/Gemini-CLI/Antigravity envelope - Kiro -- Kursori +- Cursor -Käännöksissä käytetään keskitinmuotona**OpenAI-muotoa**— kaikki konversiot menevät OpenAI:n kautta välimuotona:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Käännökset valitaan dynaamisesti lähteen hyötykuorman muodon ja toimittajan kohdemuodon perusteella. +Additional processing layers in the translation pipeline: -Muut käsittelytasot käännösputkessa: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Vastausten desinfiointi**– Poistaa standardista poikkeavat kentät OpenAI-muotoisista vastauksista (sekä suoratoistosta että ei-suoratoistosta) varmistaakseen tiukan SDK-yhteensopivuuden --**Roolin normalisointi**— Muuntaa "kehittäjä" → "järjestelmä" muille kuin OpenAI-kohteille; yhdistää `system` → `user` malleille, jotka hylkäävät järjestelmäroolin (GLM, ERNIE) --**Think-tunnisteen purkaminen**— Jäsentää "..." -lohkot sisällöstä "reasoning_content"-kenttään --**Strukturoitu tulos**— Muuntaa OpenAI `response_format.json_schema` Geminin `responseMimeType` + `responseSchema`.## Supported API Endpoints +## Supported API Endpoints -| Päätepiste | Muoto | Käsittelijä | -| --------------------------------------------------- | ------------------- | -------------------------------------------------------------------- | -| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | -| `POST /v1/messages` | Claude Viestit | Sama käsittelijä (tunnistettu automaattisesti) | -| `POST /v1/responses` | OpenAI-vastaukset | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/embeddings` | OpenAI Embeddings | "open-sse/handlers/embeddings.ts" | -| `HAE /v1/embeddings` | Malliluettelo | API reitti | -| `POST /v1/images/generations` | OpenAI-kuvat | `open-sse/handlers/imageGeneration.ts` | -| `GET /v1/images/generations` | Malliluettelo | API reitti | -| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Palveluntarjoajakohtainen mallin validointi | -| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Palveluntarjoajakohtainen mallin validointi | -| `POST /v1/providers/{provider}/images/generations` | OpenAI-kuvat | Palveluntarjoajakohtainen mallin validointi | -| `POST /v1/messages/count_tokens` | Claude Token Count | API reitti | -| `HAE /v1/mallit` | OpenAI-mallien luettelo | API-reitti (chat + upotus + kuva + mukautetut mallit) | -| "GET /api/models/catalog" | Luettelo | Kaikki mallit ryhmitelty tarjoajan + tyypin mukaan | -| `POST /v1beta/models/*:streamGenerateContent` | Gemini syntyperäinen | API reitti | -| `GET/PUT/DELETE /api/settings/proxy` | Välityspalvelimen kokoonpano | Verkon välityspalvelimen määritykset | -| "POST /api/settings/proxy/test" | Välityspalvelinyhteydet | Välityspalvelimen kunto/yhteystestin päätepiste | -| `GET/POST/DELETE /api/provider-models` | Palveluntarjoajan mallit | Palveluntarjoajan mallin metatietojen tausta mukautettuja ja hallittuja saatavilla olevia malleja |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -Ohituskäsittelijä (`open-sse/utils/bypassHandler.ts`) sieppaa Claude CLI:n tunnetut "heittopyynnöt" – lämmittelypingit, otsikon poiminnot ja tunnukset - ja palauttaa**väärennetyn vastauksen**kuluttamatta alkupään toimittajatunnuksia. Tämä käynnistyy vain, kun "User-Agent" sisältää "claude-cli".## Request Logger Pipeline +## Bypass Handler -Pyyntöloggeri (`open-sse/utils/requestLogger.ts`) tarjoaa 7-vaiheisen virheenkorjauslokiputken, joka on oletusarvoisesti pois käytöstä ja joka on käytössä kohdassa ENABLE_REQUEST_LOGS=true:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Tiedostot kirjoitetaan hakemistoon `/logs//` jokaista pyyntöistuntoa varten.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- Palveluntarjoajan tilin jäähtyminen ohimenevien / nopeus / todennusvirheiden vuoksi -- tilin varaosa ennen epäonnistunutta pyyntöä -- Yhdistelmämallin palautus, kun nykyisen mallin/palveluntarjoajan polku on käytetty loppuun## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- esitarkista ja päivitä yrittämällä uudelleen päivitettävien palveluntarjoajien kohdalla -- 401/403 yritä uudelleen päivitysyrityksen jälkeen ydinpolulla## 3) Stream Safety +## 2) Token Expiry -- irrotettava stream-ohjain -- käännösvirta streamin lopun huuhtelulla ja [VALMIS]-käsittelyllä -- käyttöarvion varavaihtoehto, kun palveluntarjoajan käytön metatiedot puuttuvat## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- Synkronointivirheet tulevat esiin, mutta paikallinen suoritusaika jatkuu -- ajastimessa on uudelleenyrityslogiikka, mutta säännöllinen suoritus tällä hetkellä kutsuu oletusarvoisesti yhden yrityksen synkronointia## 5) Data Integrity +## 3) Stream Safety -- SQLite-skeeman siirrot ja automaattisen päivityksen koukut käynnistyksen yhteydessä -- vanha JSON → SQLite-siirtoyhteensopivuuspolku## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Ajonaikaisen näkyvyyden lähteet: +## 4) Cloud Sync Degradation -- konsolin lokit osoitteesta "src/sse/utils/logger.ts". -- SQLiten pyyntökohtaiset käyttöaggregaatit ("usage_history", "call_logs", "proxy_logs") -- nelivaiheiset yksityiskohtaiset hyötykuorman kaappaukset SQLitessa (`request_detail_logs`), kun `settings.detailed_logs_enabled=true` -- tekstimuotoisen pyynnön tilaloki tiedostossa "log.txt" (valinnainen/compat) -- valinnaiset syvät pyyntö-/käännöslokit lokit/-kohdassa, kun ENABLE_REQUEST_LOGS=true -- hallintapaneelin käyttöpäätepisteet (`/api/usage/*`) käyttöliittymän käyttöä varten +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -Yksityiskohtainen pyyntöhyötykuormakaappaus tallentaa jopa neljä JSON-hyötykuorman vaihetta reititettyä puhelua kohden: +## 5) Data Integrity -- Asiakkaalta saatu raakapyyntö -- käännetty pyyntö todella lähetetty alkupäässä -- palveluntarjoajan vastaus rekonstruoitu JSON-muodossa; suoratoistovastaukset tiivistetään lopulliseksi yhteenvedoksi ja virran metadataksi -- OmniRouten palauttama lopullinen asiakkaan vastaus; suoratoistovastaukset tallennetaan samaan kompaktiin tiivistelmään## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- JWT-salaisuus (`JWT_SECRET`) suojaa hallintapaneelin istunnon evästeen vahvistuksen/allekirjoituksen -- Alkuperäisen salasanan käynnistys (`INITIAL_PASSWORD`) on määritettävä eksplisiittisesti ensiajoa varten -- API-avaimen HMAC-salaisuus (`API_KEY_SECRET`) suojaa luodun paikallisen API-avainmuodon -- Tarjoajan salaisuudet (API-avaimet/tunnisteet) säilyvät paikallisessa tietokannassa, ja ne tulee suojata tiedostojärjestelmätasolla -- Pilvisynkronoinnin päätepisteet perustuvat API-avaimen todennus + konetunnuksen semantiikkaan## Environment and Runtime Matrix +## Observability and Operational Signals -Koodin aktiivisesti käyttämät ympäristömuuttujat: +Runtime visibility sources: -- Sovellus/todennus: "JWT_SECRET", "INITIAL_PASSWORD" -- Tallennustila: "DATA_DIR". -- Yhteensopivan solmun toiminta: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Valinnainen tallennuskannan ohitus (Linux/macOS, kun "DATA_DIR" ei ole asetettu): "XDG_CONFIG_HOME" -- Suojaushajautus: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Kirjaaminen: "ENABLE_REQUEST_LOGS". -- Synkronointi/pilvi-URL-osoite: NEXT_PUBLIC_BASE_URL, NEXT_PUBLIC_CLOUD_URL -- Lähtevä välityspalvelin: "HTTP_PROXY", "HTTPS_PROXY", "ALL_PROXY", "NO_PROXY" ja pienet versiot -- SOCKS5-ominaisuuden liput: "ENABLE_SOCKS5_PROXY", "NEXT_PUBLIC_ENABLE_SOCKS5_PROXY" -- Alusta/ajonaikaiset apuohjelmat (ei sovelluskohtaiset asetukset): "APPDATA", "NODE_ENV", "PORTTI", "HOSTNAME"## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` ja `localDb` jakavat saman perushakemistokäytännön (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) vanhojen tiedostojen siirrolla. -2. "/api/v1/route.ts" siirtää samaan yhdistetyn luettelon rakennustyökaluun, jota "/api/v1/models" ("src/app/api/v1/models/catalog.ts") käyttää semanttisen ajautumisen välttämiseksi. -3. Pyyntöloggeri kirjoittaa täydet otsikot/runko, kun se on käytössä; käsittele lokihakemistoa arkaluontoisena. -4. Pilven toiminta riippuu oikeasta NEXT_PUBLIC_BASE_URL-osoitteesta ja pilvipäätepisteen saavutettavuudesta. -5. Hakemisto "open-sse/" julkaistaan ​​@omniroute/open-sse**npm-työtilapaketina**. Lähdekoodi tuo sen @omniroute/open-sse/...-tiedoston kautta (ratkaisi Next.js `transpilePackages`). Tämän asiakirjan tiedostopolut käyttävät edelleen hakemistonimeä `open-sse/` johdonmukaisuuden vuoksi. -6. Hallintapaneelin kaaviot käyttävät**Uudelleenkaavioita**(SVG-pohjainen) helppokäyttöisten, interaktiivisten analytiikkavisualisoinnit (mallien käyttöpalkkikaaviot, toimittajien erittelytaulukot onnistumisprosentteineen) varten. -7. E2E-testit käyttävät**Playwrightia**(`tests/e2e/`), suoritetaan komennolla "npm run test:e2e". Yksikkötesteissä käytetään**Node.js-testirunneria**(`tests/unit/`), suoritetaan komennolla "npm run test:unit". Lähdekoodi kohdassa `src/` on**TypeScript**(`.ts`/`.tsx`); `open-sse/`-työtila pysyy JavaScriptina (`.js`). -8. Asetukset-sivu on järjestetty viiteen välilehteen: Suojaus, Reititys (6 globaalia strategiaa: täytä ensin, round-robin, p2c, satunnainen, vähiten käytetty, kustannusoptimoitu), Resilience (muokattavat nopeusrajoitukset, katkaisija, käytännöt), AI (ajattelubudjetti, järjestelmäkehote, kehote välimuisti), Advanced (välityspalvelin).## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- Koonti lähteestä: `npm run build` -- Build Docker -kuva: `docker build -t omniroute .` -- Aloita huolto ja varmista: -- "HAE /api/settings". -- "GET /api/v1/models". -- CLI-kohteen perus-URL-osoitteen tulee olla "http://:20128/v1", kun PORT=20128 +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` +- `GET /api/v1/models` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/fi/docs/FEATURES.md b/docs/i18n/fi/docs/FEATURES.md index cf0668989f..cc7e406c0b 100644 --- a/docs/i18n/fi/docs/FEATURES.md +++ b/docs/i18n/fi/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Visuaalinen opas OmniRoute-hallintapaneelin jokaiseen osioon.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Hallinnoi AI-palveluntarjoajan yhteyksiä: OAuth-palveluntarjoajat (Claude Code, Codex, Gemini CLI), API-avaintoimittajat (Groq, DeepSeek, OpenRouter) ja ilmaiset palveluntarjoajat (Qoder, Qwen, Kiro). Kiro-tilit sisältävät luottosaldon seurannan – jäljellä olevat saldot, kokonaisrahoitus ja uusimispäivä näkyvät kohdassa Dashboard → Käyttö.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Luo mallin reitityskomboja kuudella strategialla: prioriteetti, painotettu, kiertävä, satunnainen, vähiten käytetty ja kustannusoptimoitu. Jokainen yhdistelmä ketjuttaa useita malleja automaattisilla varauksilla ja sisältää nopeat mallit ja valmiustarkistukset.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Kattava käyttöanalytiikka tunnuksen kulutuksella, kustannusarvioilla, aktiivisuuslämpökartoilla, viikoittaisilla jakelukaavioilla ja palveluntarjoajakohtaisilla erittelyillä.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Reaaliaikainen seuranta: käyttöaika, muisti, versio, latenssiprosenttipisteet (p50/p95/p99), välimuistitilastot ja palveluntarjoajan katkaisijan tilat.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Neljä tilaa API-käännösten virheenkorjaukseen:**Playground**(muodonmuunnin),**Chat Tester**(livepyynnöt),**Test Bench**(erätestit) ja**Live Monitor**(reaaliaikainen suoratoisto).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Testaa mitä tahansa mallia suoraan kojelaudalta. Valitse palveluntarjoaja, malli ja päätepiste, kirjoita kehotteita Monaco Editorilla, suoratoista vastaukset reaaliajassa, keskeytä kesken stream ja tarkastele ajoitusmittauksia.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Muokattavat väriteemat koko kojelautaan. Valitse 7 esiasetetusta väristä (koralli, sininen, punainen, vihreä, violetti, oranssi, syaani) tai luo mukautettu teema valitsemalla mikä tahansa kuusioväri. Tukee vaaleaa, tummaa ja järjestelmätilaa.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Kattava asetuspaneeli välilehdillä: +Comprehensive settings panel with tabs: --**Yleistä**- Järjestelmän tallennus, varmuuskopioiden hallinta (vienti/tuonti tietokanta) -**Ulkoasu**- Teeman valitsin (tumma/vaalea/järjestelmä), väriteeman esiasetukset ja mukautetut värit, terveyslokin näkyvyys, sivupalkin kohteiden näkyvyyden säätimet -**Turvallisuus**— API-päätepisteiden suojaus, mukautetun palveluntarjoajan esto, IP-suodatus, istuntotiedot -**Reititys**— Mallin aliakset, taustatehtävän huononeminen -**Kestävyys**— Hintarajoituksen pysyvyys, katkaisijan viritys, estettyjen tilien automaattinen poistaminen käytöstä, palveluntarjoajan vanhenemisen valvonta -**Lisäasetukset**— Kokoonpanon ohitukset, määrityksen kirjausketju, varatilan heikkenemistila![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Yhden napsautuksen konfigurointi AI-koodaustyökaluille: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor ja Factory Droid. Sisältää automaattisen konfiguroinnin käyttöönotto/nollaus, yhteysprofiilit ja mallikartoituksen.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Kojelauta CLI-agenttien löytämiseen ja hallintaan. Näyttää 14 sisäänrakennetun agentin (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) ruudukon, jossa on: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Asennustila**— Asennettu / Ei löydy versiontunnistuksen kanssa -**Protokollamerkit**— stdio, HTTP jne. -**Muokatut agentit**— Rekisteröi mikä tahansa CLI-työkalu lomakkeella (nimi, binaari, versiokomento, spawn args) -**CLI-sormenjälkien vastaavuus**– Palveluntarjoajakohtainen kytkin vastaamaan alkuperäisten CLI-pyyntöjen allekirjoituksia, mikä vähentää eston riskiä ja säilyttää välityspalvelimen IP-osoitteen--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Luo kuvia, videoita ja musiikkia kojelaudalta. Tukee OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open ja MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Reaaliaikainen pyyntöjen kirjaaminen suodatuksella palveluntarjoajan, mallin, tilin ja API-avaimen mukaan. Näyttää tilakoodit, tunnuksen käytön, viiveen ja vastaustiedot.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Yhdistetty API-päätepisteesi ominaisuuksien erittelyllä: Chat Completions, Responses API, upotukset, kuvan luominen, uudelleensijoitus, äänen transkriptio, tekstistä puheeksi, moderaatiot ja rekisteröidyt API-avaimet. Cloudflare Quick Tunnel -integraatio ja pilvivälityspalvelintuki etäkäyttöä varten.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -Luo, laajenna ja peruuta API-avaimia. Jokainen avain voidaan rajoittaa tiettyihin malleihin/palveluntarjoajiin, joilla on täydet käyttöoikeudet tai vain lukuoikeudet. Visuaalinen avainten hallinta käytön seurannalla.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Hallinnollinen toimintojen seuranta suodatuksella toimintotyypin, toimijan, kohteen, IP-osoitteen ja aikaleiman mukaan. Täydellinen tietoturvatapahtumahistoria.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Native Electron -työpöytäsovellus Windowsille, macOS:lle ja Linuxille. Suorita OmniRoute itsenäisenä sovelluksena, jossa on järjestelmälokeron integrointi, offline-tuki, automaattinen päivitys ja asennus yhdellä napsautuksella. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Tärkeimmät ominaisuudet: +Key features: -- Palvelimen valmiuskysely (ei tyhjää näyttöä kylmäkäynnistyksen yhteydessä) -- Järjestelmälokero portinhallinnan kanssa -- Sisällön suojauskäytäntö -- Yksiosainen lukko -- Automaattinen päivitys uudelleenkäynnistyksen yhteydessä -- Alustan ehdollinen käyttöliittymä (macOS-liikennevalot, Windowsin/Linuxin oletusotsikkopalkki) -- Hardened Electron build -pakkaus – itsenäisen nipun symlinkoidut "solmumoduulit" tunnistetaan ja hylätään ennen pakkausta, mikä estää ajonaikaisen riippuvuuden rakennuskoneesta (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Katso täydelliset asiakirjat osoitteesta [`electron/README.md`](../electron/README.md). +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/fi/docs/TROUBLESHOOTING.md b/docs/i18n/fi/docs/TROUBLESHOOTING.md index d301f4f000..c2fbb68c02 100644 --- a/docs/i18n/fi/docs/TROUBLESHOOTING.md +++ b/docs/i18n/fi/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -OmniRouten yleisiä ongelmia ja ratkaisuja.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Ongelma | Ratkaisu | -| ---------------------------------- | ------------------------------------------------------------------------------- | --- | -| Ensimmäinen kirjautuminen ei toimi | Aseta 'INITIAL_PASSWORD' .env:ssä (ei kovakoodattua oletusarvoa) | -| Kojelauta avautuu väärään porttiin | Aseta `PORT=20128` ja `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Ei pyyntölokeja kohdassa "lokit/" | Aseta ENABLE_REQUEST_LOGS=true | -| EACCES: lupa evätty | Aseta "DATA_DIR=/polku/kirjoitettavaan/hakemistoon" ohittaaksesi "~/.omniroute" | -| Reititysstrategia ei tallennu | Päivitys versioon 1.4.11+ (Zod-skeeman korjaus asetusten pysyvyyttä varten) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Syy:**Palveluntarjoajan kiintiö käytetty. +**Cause:** Provider quota exhausted. -**Korjaa:** +**Fix:** -1. Tarkista kojelaudan kiintiöiden seuranta -2. Käytä yhdistelmää varatasoilla -3. Vaihda halvempaan/ilmaiseen tasoon### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Syy:**Tilauskiintiö käytetty. +### Rate Limiting -**Korjaa:** +**Cause:** Subscription quota exhausted. -- Lisää vara: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Käytä GLM/MiniMaxia halvana varmuuskopiona### OAuth Token Expired +**Fix:** -OmniRoute päivittää tunnukset automaattisesti. Jos ongelmat jatkuvat: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Kojelauta → Palveluntarjoaja → Yhdistä uudelleen -2. Poista ja lisää palveluntarjoajan yhteys uudelleen--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Varmista, että BASE_URL osoittaa käynnissä olevaan esiintymääsi (esim. http://localhost:20128) -2. Varmista, että CLOUD_URL-osoite osoittaa pilvipäätepisteeseesi (esim. https://omniroute.dev). -3. Pidä NEXT*PUBLIC*\*-arvot kohdakkain palvelinpuolen arvojen kanssa### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Oire:**"Odottamaton tunnus "d"..." pilvipäätepisteessä ei-suoratoistopuheluille. +### Cloud `stream=false` Returns 500 -**Syy:**Upstream palauttaa SSE-hyötykuorman, kun asiakas odottaa JSONia. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Ratkaisu:**Käytä "stream=true" pilvisuorapuheluissa. Paikallinen suoritusaika sisältää SSE→JSON-varavaihtoehdon.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Luo uusi avain paikallisesta hallintapaneelista (`/api/keys`) -2. Suorita pilvisynkronointi: Ota pilvi käyttöön → Synkronoi nyt -3. Vanhat/synkronoimattomat avaimet voivat edelleen palauttaa 401:n pilvessä--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Tarkista ajonaikaiset kentät: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. Kannettava tila: käytä kuvakohdetta "runner-cli" (niputetut CLI:t) -3. Isäntäliitostila: aseta CLI_EXTRA_PATHS ja liitä isäntäalustahakemisto vain luku -muotoiseksi -4. Jos "installed=true" ja "runnable=false": binaari löytyi, mutta kuntotarkastus epäonnistui### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Tarkista käyttötilastot kohdassa Dashboard → Usage -2. Vaihda ensisijaiseksi malliksi GLM/MiniMax -3. Käytä ilmaista tasoa (Gemini CLI, Qoder) ei-kriittisiin tehtäviin -4. Aseta kustannusbudjetit API-avainta kohti: Dashboard → API Keys → Budget--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Aseta ENABLE_REQUEST_LOGS=true .env-tiedostoosi. Lokit näkyvät lokit/hakemistossa.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Päätila: `${DATA_DIR}/storage.sqlite` (palveluntarjoajat, yhdistelmät, aliakset, avaimet, asetukset) -- Käyttö: SQLite-taulukot tiedostossa "storage.sqlite" ("usage_history", "call_logs", "proxy_logs") + valinnainen "${DATA_DIR}/log.txt" ja "${DATA_DIR}/call_logs/" -- Pyydä lokeja: `/logs/...` (kun `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Kun palveluntarjoajan katkaisija on AUKI, pyynnöt estetään, kunnes jäähdytys päättyy. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Korjaa:** +**Fix:** -1. Siirry kohtaan**Käyttöpaneeli → Asetukset → Resilience** -2. Tarkista asianomaisen palveluntarjoajan katkaisijakortti -3. Napsauta**Nollaa kaikki**tyhjentääksesi kaikki katkaisijat tai odota jäähdytysajan päättymistä -4. Varmista, että palveluntarjoaja on todella saatavilla, ennen kuin nollaat### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Jos palveluntarjoaja siirtyy toistuvasti OPEN-tilaan: +### Provider keeps tripping the circuit breaker -1. Tarkista vikakuvio kohdasta**Dashboard → Health → Provider Health** -2. Siirry kohtaan**Settings → Resilience → Provider Profiles**ja nosta vikakynnystä. -3. Tarkista, onko palveluntarjoaja muuttanut API-rajoja tai vaatiiko todennuksen uudelleen -4. Tarkista viiveen telemetria — korkea latenssi voi aiheuttaa aikakatkaisuun perustuvia virheitä--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Varmista, että käytät oikeaa etuliitettä: "deepgram/nova-3" tai "assemblyai/best" -- Varmista, että palveluntarjoaja on yhdistetty kohdassa**Dashboard → Providers**### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Tarkista tuetut äänimuodot: "mp3", "wav", "m4a", "flac", "ogg", "webm" -- Varmista, että tiedostokoko on palveluntarjoajan rajoissa (yleensä < 25 Mt) -- Tarkista palveluntarjoajan API-avaimen voimassaolo toimittajakortista--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Käytä**Käyttöpaneeli → Kääntäjä**muotojen käännösongelmien korjaamiseen: +Use **Dashboard → Translator** to debug format translation issues: -| Tila | Milloin käyttää | -| ------------------------- | ------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Leikkikenttä** | Vertaa syöttö-/tulostusmuotoja vierekkäin – liitä epäonnistunut pyyntö nähdäksesi, miten se käännetään | -| **Pikaviestien testaaja** | Lähetä reaaliaikaisia ​​viestejä ja tarkasta koko pyynnön/vastauksen hyötykuorma, mukaan lukien otsikot | -| **Testipenkki** | Suorita erätestejä muotoyhdistelmille selvittääksesi, mitkä käännökset ovat rikki | -| **Live Monitor** | Tarkkaile reaaliaikaista pyyntövirtaa havaitaksesi ajoittaiset käännösongelmat | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Ajattelevat tunnisteet eivät näy**— Tarkista, tukeeko kohdetoimittaja ajattelua ja ajattelun budjettiasetusta -**Työkalukutsujen pudottaminen**— Jotkin muotokäännökset voivat poistaa ei-tuetut kentät. vahvista leikkikenttätilassa -**Järjestelmäkehote puuttuu**— Claude ja Gemini kahvajärjestelmä kehottaa eri tavalla; tarkista käännöstulos -**SDK palauttaa raakamerkkijonon objektin sijaan**— Korjattu versiossa 1.1.0: vastauspuhdistin poistaa nyt standardista poikkeavat kentät ("x_groq", "usage_breakdown" jne.), jotka aiheuttavat OpenAI SDK Pydantic -tarkistusvirheitä -**GLM/ERNIE hylkää "järjestelmän" roolin**- Korjattu versiossa 1.1.0: roolin normalisoija yhdistää automaattisesti järjestelmäviestit käyttäjäviesteiksi yhteensopimattomissa malleissa -**"kehittäjäroolia" ei tunnistettu**- Korjattu versiossa 1.1.0: muunnetaan automaattisesti "järjestelmäksi" muille kuin OpenAI-palveluntarjoajille -**`json_schema` ei toimi Geminin kanssa**— Korjattu versiossa 1.1.0: `response_format` muunnetaan nyt Geminin `responseMimeType` + `responseSchema` -muotoon.--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- Automaattinen nopeusrajoitus koskee vain API-avainten toimittajia (ei OAuth-tilausta) -- Varmista, että**Asetukset → Resilienssi → Palveluntarjoajan profiilit**on automaattinen rajoitus käytössä -- Tarkista, palauttaako palveluntarjoaja "429"-tilakoodit tai "Retry-After"-otsikot### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -Palveluntarjoajan profiilit tukevat näitä asetuksia: +### Tuning exponential backoff --**Perusviive**— Ensimmäinen odotusaika ensimmäisen epäonnistumisen jälkeen (oletus: 1 s) -**Maksimiviive**- Odotusajan enimmäisraja (oletus: 30 s) -**Kerroin**— Kuinka paljon viivettä lisätään peräkkäistä vikaa kohti (oletus: 2x)### Anti-thundering herd +Provider profiles support these settings: -Kun monet samanaikaiset pyynnöt osuvat nopeusrajoitettuun palveluntarjoajaan, OmniRoute käyttää mutex + automaattista nopeuden rajoitusta sarjoittamaan pyynnöt ja estämään peräkkäiset epäonnistumiset. Tämä on automaattinen API-avainten tarjoajille.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Jotkut OmniRouten käyttäjät sijoittavat yhdyskäytävän RAG- tai agenttipinojen eteen. Näissä asetuksissa on tavallista nähdä outo kuvio: OmniRoute näyttää terveeltä (palveluntarjoajat valmiina, reititysprofiilit kunnossa, ei nopeusrajoitushälytyksiä), mutta lopullinen vastaus on silti väärä. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -Käytännössä nämä tapaukset tulevat yleensä loppupään RAG-putkistosta, eivät itse yhdyskäytävästä. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Jos haluat jaetun sanaston kuvaamaan näitä vikoja, voit käyttää WFGY ProblemMapia, ulkoista MIT-lisenssitekstiresurssia, joka määrittelee kuusitoista toistuvaa RAG/LLM-vikamallia. Korkealla tasolla se kattaa: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- haun ajautuminen ja rikotut kontekstin rajat -- tyhjät tai vanhentuneet indeksit ja vektorivarastot -- upottaminen vs. semanttinen yhteensopivuus -- Nopeat kokoonpano- ja kontekstiikkuna-ongelmat -- logiikka romahtaa ja liian itsevarmat vastaukset -- pitkän ketjun ja agenttien koordinaatiohäiriöt -- monen agentin muisti ja roolien siirtyminen -- käyttöönotto- ja käynnistystilausongelmat +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -Idea on yksinkertainen: +The idea is simple: -1. Kun tutkit huonoa vastausta, tallenna: - - käyttäjän tehtävä ja pyyntö - - reitti- tai tarjoajayhdistelmä OmniRoutessa - - mikä tahansa loppupäässä käytetty RAG-konteksti (haettu asiakirjat, työkalukutsut jne.) -2. Kartoita tapahtuma yhteen tai kahteen WFGY-ongelmakarttanumeroon (`No.1` … `No.16`). -3. Tallenna numero omaan kojelautaan, runbookiin tai tapahtumaseurantaan OmniRoute-lokien viereen. -4. Käytä vastaavaa WFGY-sivua päättääksesi, onko sinun muutettava RAG-pinoa, noutajaa tai reititysstrategiaa. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -Koko teksti ja konkreettiset reseptit löytyvät täältä (MIT-lisenssi, vain teksti): +Full text and concrete recipes live here (MIT license, text only): [WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -Voit jättää tämän osion huomioimatta, jos et käytä RAG- tai agenttiputkia OmniRouten takana.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**GitHub-ongelmat**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Arkkitehtuuri**: Katso sisäiset tiedot osoitteesta [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) -**API-viite**: Katso [`docs/API_REFERENCE.md`](API_REFERENCE.md) kaikista päätepisteistä -**Health Dashboard**: Tarkista järjestelmän reaaliaikainen tila kohdasta**Dashboard → Health** -**Kääntäjä**: Käytä**Käyttöpaneeli → Kääntäjä**muotoongelmien korjaamiseen +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/fi/llm.txt b/docs/i18n/fi/llm.txt new file mode 100644 index 0000000000..e0b98c94a8 --- /dev/null +++ b/docs/i18n/fi/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Suomi) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Yleiskatsaus + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Turvallisuus +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/fr/README.md b/docs/i18n/fr/README.md index 5399e19643..362bdd077a 100644 --- a/docs/i18n/fr/README.md +++ b/docs/i18n/fr/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Votre proxy API universel : un point de terminaison, plus de 60 fournisseurs, aucun temps d'arrêt. Désormais avec**Serveur MCP (25 outils)**,**Protocole A2A**,**Systèmes de mémoire/compétences**et**Application de bureau Electron**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Fin de discussions • Intégrations • Génération d'images • Vidéo • Musique • Audio • Reclassement •**Recherche sur le Web**• Serveur MCP • Protocole A2A • 100 % TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Votre proxy API universel : un point de terminaison, plus de 60 fournisseurs, [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Site Web](https://omniroute.online) • [🚀 Démarrage rapide](#-démarrage rapide) • [💡 Fonctionnalités](#-fonctionnalités-clés) • [📖 Documents](#-documentation) • [💰 Tarification](#-tarification-en un coup d'œil) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Disponible en :**🇺🇸 [anglais](README.md) | 🇧🇷 [Português (Brésil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italien](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonésie](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Pays-Bas](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Philippin](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,29 +60,30 @@ _Votre proxy API universel : un point de terminaison, plus de 60 fournisseurs, ## 📸 Dashboard Preview - +
+Click to see dashboard screenshots -Cliquez pour voir les captures d'écran du tableau de bord +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | -| Pages | Capture d'écran | -| -------------------------- | ---------------------------------------------------------- | ---------- | -| **Fournisseurs** | ![Fournisseurs](docs/screenshots/01-providers.png) | -| **Combinaisons** | ![Combos](docs/screenshots/02-combos.png) | -| **Analyses** | ![Analyses](docs/screenshots/03-analytics.png) | -| **Santé** | ![Santé](docs/screenshots/04-health.png) | -| **Traducteur** | ![Traducteur](docs/screenshots/05-translator.png) | -| **Paramètres** | ![Paramètres](docs/screenshots/06-settings.png) | -| **Outils CLI** | ![Outils CLI](docs/screenshots/07-cli-tools.png) | -| **Journaux d'utilisation** | ![Utilisation](docs/screenshots/08-usage.png) | -| **Points de terminaison** | ![Points de terminaison](docs/screenshots/09-endpoint.png) |
| + --- ### 🤖 Free AI Provider for your favorite coding agents -_Connectez n'importe quel outil IDE ou CLI alimenté par l'IA via OmniRoute — une passerelle API gratuite pour un codage illimité._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ - + @@ -124,483 +132,557 @@ _Connectez n'importe quel outil IDE ou CLI alimenté par l'IA via OmniRoute —
@@ -89,28 +97,28 @@ _Connectez n'importe quel outil IDE ou CLI alimenté par l'IA via OmniRoute — NanoBot
NanoBot

- ⭐ 20,9K + ⭐ 20.9K
PicoClaw
- PicoGriffe + PicoClaw

- ⭐ 14,6K + ⭐ 14.6K
ZeroClaw
- ZéroClaw + ZeroClaw

- ⭐ 9,9K + ⭐ 9.9K
IronClaw
- Griffe de Fer + IronClaw

- ⭐ 2,1K + ⭐ 2.1K
Codex CLI
- CLI Codex + Codex CLI

- ⭐ 60,8K + ⭐ 60.8K
Claude Code
Claude Code

- ⭐ 67,3K + ⭐ 67.3K
Gemini CLI
- CLI Gemini + Gemini CLI

- ⭐ 94,7K + ⭐ 94.7K
- Code kilo
- Code kilo + Kilo Code
+ Kilo Code

- ⭐ 15,5K + ⭐ 15.5K
-📡 Tous les agents se connectent via http://localhost:20128/v1 ou http://cloud.omniroute.online/v1 — une configuration, des modèles et un quota illimités--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Arrêtez de gaspiller de l'argent et d'atteindre vos limites :** +**Stop wasting money and hitting limits:** -- Le quota d'abonnement expire chaque mois sans être utilisé -- Les limites de débit vous empêchent de coder -- API coûteuses (20-50 $/mois par fournisseur) -- Commutation manuelle entre les fournisseurs +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute résout ce problème :** +**OmniRoute solves this:** -- ✅**Maximiser les abonnements**- Suivez le quota, utilisez chaque bit avant la réinitialisation -- ✅**Repli automatique**- Abonnement → Clé API → Pas cher → Gratuit, aucun temps d'arrêt -- ✅**Multi-compte**- Round-robin entre les comptes par fournisseur -- ✅**Universel**- Fonctionne avec Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, n'importe quel outil CLI--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Rejoignez notre communauté !**[Groupe WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Obtenez de l'aide, partagez des conseils et restez informé. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Site Internet**: [omniroute.online](https://omniroute.online) -**GitHub** : [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Problèmes** : [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp** : [Groupe communautaire](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Contribuer** : voir [CONTRIBUTING.md](CONTRIBUTING.md), ouvrir un PR ou choisir un « bon premier numéro » -**Projet original** : [9router par decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Lors de l'ouverture d'un ticket, veuillez exécuter la commande system-info et joindre le fichier généré :```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Cela génère un `system-info.txt` avec votre version de Node.js, la version d'OmniRoute, les détails du système d'exploitation, les outils CLI installés (qoder, gemini, claude, codex, antigravity, droid, etc.), l'état Docker/PM2 et les packages système — tout ce dont nous avons besoin pour reproduire rapidement votre problème. Joignez le fichier directement à votre problème GitHub.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Tous les développeurs utilisant des outils d'IA sont confrontés quotidiennement à ces problèmes.**OmniRoute a été conçu pour tous les résoudre : des dépassements de coûts aux blocages régionaux, des flux OAuth interrompus aux opérations de protocole et à l'observabilité de l'entreprise. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. - -💸 1. "Je paie un abonnement coûteux mais je suis quand même interrompu par des limites" +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Les développeurs paient entre 20 et 200 $/mois pour Claude Pro, Codex Pro ou GitHub Copilot. Même payant, le quota est plafonné : 5 heures d'utilisation, limites hebdomadaires ou limites de tarif à la minute. En cours de session de codage, le fournisseur ne répond plus et le développeur perd en fluidité et en productivité. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Comment OmniRoute le résout :** +**How OmniRoute solves it:** --**Smart 4-Tier Fallback**— Si le quota d'abonnement est épuisé, redirige automatiquement vers la clé API → Pas cher → Gratuit sans intervention manuelle --**Suivi des limites du fournisseur**— Actualisation des instantanés de quotas mis en cache selon une planification côté serveur (par défaut `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) avec actualisation manuelle disponible dans l'interface utilisateur --**Support multi-comptes**— Plusieurs comptes par fournisseur avec tourniquet automatique — lorsqu'un compte est épuisé, passe au suivant --**Combos personnalisés**— Chaînes de secours personnalisables avec 9 stratégies d'équilibrage (priorité, pondérée, remplissage en premier, round-robin, P2C, aléatoire, la moins utilisée, optimisée en termes de coût, strictement aléatoire) --**Codex Business Quotas**— Surveillance des quotas d'espace de travail Business/Équipe directement dans le tableau de bord
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard - -🔌 2. "Je dois utiliser plusieurs fournisseurs mais chacun a une API différente" + -OpenAI utilise un format, Claude (Anthropic) en utilise un autre, Gemini encore un autre. Si un développeur souhaite tester des modèles de différents fournisseurs ou utiliser un modèle de secours entre eux, il doit reconfigurer les SDK, modifier les points de terminaison et gérer les formats incompatibles. Les fournisseurs personnalisés (FriendLI, NIM) ont des points de terminaison de modèle non standard. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Comment OmniRoute le résout :** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Unified Endpoint**— Un seul « http://localhost:20128/v1 » sert de proxy pour plus de 60 fournisseurs. --**Traduction de format**— Automatique et transparente : OpenAI ↔ Claude ↔ Gemini ↔ API Responses --**Response Sanitization**— Supprime les champs non standard (`x_groq`, `usage_breakdown`, `service_tier`) qui cassent OpenAI SDK v1.83+ --**Role Normalization**— Convertit « développeur » → « système » pour les fournisseurs non OpenAI ; `système` → `utilisateur` pour GLM/ERNIE --**Think Tag Extraction**— Extrait les blocs `` de modèles comme DeepSeek R1 dans un `reasoning_content` standardisé --**Sortie structurée pour Gemini**— Conversion automatique `json_schema` → `responseMimeType`/`responseSchema` --**`stream` est par défaut `false`**— S'aligne sur les spécifications OpenAI, évitant ainsi le SSE inattendu dans les SDK Python/Rust/Go
+**How OmniRoute solves it:** - -🌐 3. "Mon fournisseur d'IA bloque ma région/mon pays" +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Des fournisseurs comme OpenAI/Codex bloquent l’accès depuis certaines régions géographiques. Les utilisateurs obtiennent des erreurs telles que « unsupported_country_region_territory » lors des connexions OAuth et API. Ceci est particulièrement frustrant pour les développeurs des pays en développement. + -**Comment OmniRoute le résout :** +
+🌐 3. "My AI provider blocks my region/country" --**Configuration proxy à 3 niveaux**— Proxy configurable à 3 niveaux : global (tout le trafic), par fournisseur (un seul fournisseur) et par connexion/clé --**Badges proxy à code couleur**— Indicateurs visuels : 🟢 proxy global, 🟡 proxy fournisseur, 🔵 proxy de connexion, affichant toujours l'adresse IP --**Échange de jetons OAuth via proxy**— Le flux OAuth passe également par le proxy, résolvant `unsupported_country_region_territory` --**Tests de connexion via proxy**— Les tests de connexion utilisent le proxy configuré (plus de contournement direct) --**Support SOCKS5**— Prise en charge complète du proxy SOCKS5 pour le routage sortant --**TLS Fingerprint Spoofing**— Empreinte digitale TLS de type navigateur via `wreq-js` pour contourner la détection des robots --**🔏 Correspondance d'empreintes digitales CLI**— Réorganise les en-têtes et les champs de corps pour qu'ils correspondent aux signatures binaires CLI natives, réduisant ainsi considérablement le risque de signalement de compte. L'adresse IP du proxy est préservée : vous bénéficiez simultanément du masquage furtif**et**IP
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. - -🆓 4. "Je veux utiliser l'IA pour coder mais je n'ai pas d'argent" +**How OmniRoute solves it:** -Tout le monde ne peut pas payer entre 20 et 200 $/mois pour des abonnements à l’IA. Les étudiants, les développeurs des pays émergents, les amateurs et les indépendants doivent avoir accès à des modèles de qualité à un coût nul. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Comment OmniRoute le résout :** + --**Fournisseurs gratuits intégrés**— Prise en charge native des fournisseurs 100 % gratuits : Qoder (5 modèles illimités via OAuth : kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 modèles illimités : qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + ID AWS Builder gratuits), Gemini CLI (180 000 jetons/mois gratuits) --**Ollama Cloud**— Modèles Ollama hébergés dans le cloud sur « api.ollama.com » avec niveau gratuit « Utilisation légère » ; utilisez le préfixe `ollamacloud/` --**Combos gratuits uniquement**— Chaîne `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = 0 $/mois sans temps d'arrêt --**NVIDIA NIM Free Access**— ~ 40 RPM d'accès gratuit pour toujours à plus de 70 modèles sur build.nvidia.com (passage des crédits aux limites de débit pures) --**Stratégie d'optimisation des coûts**— Stratégie de routage qui choisit automatiquement le fournisseur disponible le moins cher +
+🆓 4. "I want to use AI for coding but I have no money" - -🔒 5. "Je dois protéger ma passerelle IA contre tout accès non autorisé" +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -Lors de l'exposition d'une passerelle IA au réseau (LAN, VPS, Docker), toute personne possédant l'adresse peut consommer les jetons/quota du développeur. Sans protection, les API sont vulnérables aux utilisations abusives, aux injections rapides et aux abus. +**How OmniRoute solves it:** -**Comment OmniRoute le résout :** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**Gestion des clés API**— Génération, rotation et portée par fournisseur avec une page dédiée `/dashboard/api-manager` --**Autorisations au niveau du modèle**— Restreindre les clés API à des modèles spécifiques (`openai/*`, modèles génériques), avec la bascule Autoriser tout/Restreindre --**API Endpoint Protection**— Exiger une clé pour `/v1/models` et bloquer des fournisseurs spécifiques de la liste --**Auth Guard + Protection CSRF**— Toutes les routes du tableau de bord protégées avec le middleware `withAuth` + les jetons CSRF --**Rate Limiter**— Limitation du débit par IP avec fenêtres configurables --**Filtrage IP** – Liste autorisée/liste de blocage pour le contrôle d'accès --**Prompt Injection Guard**— Nettoyage contre les modèles d'invite malveillants --**Chiffrement AES-256-GCM**— Informations d'identification chiffrées au repos
+ - -🛑 6. "Mon fournisseur est tombé en panne et j'ai perdu mon flux de codage" +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -Les fournisseurs d’IA peuvent devenir instables, renvoyer des erreurs 5xx ou atteindre des limites de débit temporaires. Si un développeur dépend d'un seul fournisseur, il est interrompu. Sans disjoncteurs, des tentatives répétées peuvent faire planter l’application. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Comment OmniRoute le résout :** +**How OmniRoute solves it:** --**Disjoncteur par modèle**— Ouverture/fermeture automatique avec seuils et temps de recharge configurables (Fermé/Ouvert/Semi-ouvert), limités par modèle pour éviter les blocages en cascade --**Exponential Backoff**— Délais progressifs entre les tentatives --**Anti-Thundering Herd**— Protection mutex + sémaphore contre les tempêtes de nouvelles tentatives simultanées --**Chaînes de secours combinées**— Si le fournisseur principal échoue, passe automatiquement à travers la chaîne sans intervention --**Combo Circuit Breaker** – Désactive automatiquement les fournisseurs défaillants au sein d'une chaîne combo --**Tableau de bord de santé**— Surveillance de la disponibilité, états des disjoncteurs, verrouillages, statistiques du cache, latence p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest - -🔧 7. "La configuration de chaque outil d'IA est fastidieuse et répétitive" + -Les développeurs utilisent Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Chaque outil nécessite une configuration différente (point de terminaison API, clé, modèle). La reconfiguration lors du changement de fournisseur ou de modèle est une perte de temps. +
+🛑 6. "My provider went down and I lost my coding flow" -**Comment OmniRoute le résout :** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**CLI Tools Dashboard**— Page dédiée avec configuration en un clic pour Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline --**GitHub Copilot Config Generator**— Génère `chatLanguageModels.json` pour VS Code avec sélection groupée de modèles --**Assistant d'intégration**— Configuration guidée en 4 étapes pour les nouveaux utilisateurs --**Un point de terminaison, tous les modèles**— Configurez `http://localhost:20128/v1` une fois, accédez à plus de 60 fournisseurs
+**How OmniRoute solves it:** - -🔑 8. "Gérer les jetons OAuth de plusieurs fournisseurs est un enfer" +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot — tous utilisent OAuth 2.0 avec des jetons expirant. Les développeurs doivent se réauthentifier constamment, gérer « client_secret manquant », « redirect_uri_mismatch » et les échecs sur les serveurs distants. OAuth sur LAN/VPS est particulièrement problématique. + -**Comment OmniRoute le résout :** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Actualisation automatique des jetons** : les jetons OAuth sont actualisés en arrière-plan avant leur expiration. --**OAuth 2.0 (PKCE) intégré**— Flux automatique pour Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder --**Multi-Account OAuth**— Plusieurs comptes par fournisseur via l'extraction de jetons JWT/ID --**OAuth LAN/Remote Fix**— Détection d'adresse IP privée pour `redirect_uri` + mode URL manuel pour les serveurs distants --**OAuth derrière Nginx**— Utilise `window.location.origin` pour la compatibilité du proxy inverse --**Guide OAuth à distance**— Guide étape par étape pour les informations d'identification Google Cloud sur VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. - -📊 9. "Je ne sais pas combien je dépense ni où" +**How OmniRoute solves it:** -Les développeurs utilisent plusieurs fournisseurs payants mais n'ont pas de vue unifiée des dépenses. Chaque fournisseur dispose de son propre tableau de bord de facturation, mais il n'existe pas de vue consolidée. Les coûts inattendus peuvent s’accumuler. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Comment OmniRoute le résout :** + --**Cost Analytics Dashboard**— Suivi des coûts par jeton et gestion du budget par fournisseur --**Limites budgétaires par niveau**— Plafond de dépenses par niveau qui déclenche un repli automatique --**Configuration de tarification par modèle**— Prix configurables par modèle --**Statistiques d'utilisation par clé API**— Nombre de demandes et horodatage de la dernière utilisation par clé --**Tableau de bord Analytics**— Cartes statistiques, tableau d'utilisation du modèle, tableau des fournisseurs avec taux de réussite et latence +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" - -🐛 10. "Je ne peux pas diagnostiquer les erreurs et les problèmes dans les appels IA" +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Lorsqu'un appel échoue, le développeur ne sait pas s'il s'agit d'une limite de débit, d'un jeton expiré, d'un format incorrect ou d'une erreur du fournisseur. Journaux fragmentés sur différents terminaux. Sans observabilité, le débogage est un essai et une erreur. +**How OmniRoute solves it:** -**Comment OmniRoute le résout :** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Tableau de bord des journaux unifiés**— 4 onglets : journaux de requêtes, journaux proxy, journaux d'audit, console --**Console Log Viewer**— Visualiseur de style terminal en temps réel avec niveaux de code couleur, défilement automatique, recherche, filtre --**Journaux du proxy SQLite**— Journaux persistants qui survivent aux redémarrages du serveur --**Translator Playground**— 4 modes de débogage : Playground (traduction de format), Chat Tester (aller-retour), Test Bench (batch), Live Monitor (temps réel) --**Demande de télémétrie**— latence p50/p95/p99 + traçage X-Request-Id --**Journalisation basée sur des fichiers avec rotation**— Les journaux d'applications alternent en fonction de la taille, des jours de conservation et du nombre d'archives ; les artefacts du journal des appels alternent en fonction des jours de conservation et du nombre de fichiers --**Rapport d'informations système**— `npm run system-info` génère `system-info.txt` avec votre environnement complet (version Node, version OmniRoute, système d'exploitation, outils CLI, statut Docker/PM2). Joignez-le lorsque vous signalez des problèmes pour un tri instantané.
+ - -🏗️ 11. "Le déploiement et la maintenance de la passerelle sont complexes" +
+📊 9. "I don't know how much I'm spending or where" -L'installation, la configuration et la maintenance d'un proxy IA dans différents environnements (local, VPS, Docker, cloud) demandent beaucoup de main-d'œuvre. Des problèmes tels que les chemins codés en dur, les « EACCES » sur les répertoires, les conflits de ports et les versions multiplateformes ajoutent des frictions. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Comment OmniRoute le résout :** +**How OmniRoute solves it:** --**npm global install**— `npm install -g omniroute && omniroute` — terminé --**Docker Multi-Platform**— AMD64 + ARM64 natif (Apple Silicon, AWS Graviton, Raspberry Pi) --**Profils Docker Compose**— `base` (pas d'outils CLI) et `cli` (avec Claude Code, Codex, OpenClaw) --**Electron Desktop App**— Application native pour Windows/macOS/Linux avec barre d'état système, démarrage automatique et mode hors ligne --**Mode Split-Port**— API et tableau de bord sur des ports séparés pour des scénarios avancés (proxy inverse, réseau de conteneurs) --**Cloud Sync** – Configurez la synchronisation entre les appareils via Cloudflare Workers --**Sauvegardes DB**— Sauvegarde, restauration, exportation et importation automatiques de tous les paramètres, avec `DISABLE_SQLITE_AUTO_BACKUP` pour les sauvegardes gérées en externe
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency - -🌍 12. "L'interface est uniquement en anglais et mon équipe ne parle pas anglais" + -Les équipes des pays non anglophones, notamment en Amérique latine, en Asie et en Europe, ont du mal à utiliser des interfaces uniquement en anglais. Les barrières linguistiques réduisent l’adoption et augmentent les erreurs de configuration. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Comment OmniRoute le résout :** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Tableau de bord i18n — 30 langues**— Plus de 500 touches traduites, dont arabe, bulgare, danois, allemand, espagnol, finnois, français, hébreu, hindi, hongrois, indonésien, italien, japonais, coréen, malais, néerlandais, norvégien, polonais, portugais (PT/BR), roumain, russe, slovaque, suédois, thaï, ukrainien, vietnamien, chinois, philippin, anglais. --**Support RTL**— Prise en charge de droite à gauche pour l'arabe et l'hébreu --**README multilingues**— 30 traductions complètes de la documentation --**Sélecteur de langue**— Icône de globe dans l'en-tête pour une commutation en temps réel
+**How OmniRoute solves it:** - -🔄 13. "J'ai besoin de plus que du chat : j'ai besoin d'intégrations, d'images, d'audio" +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -L'IA ne se limite pas à la réalisation de discussions. Les développeurs doivent générer des images, transcrire l'audio, créer des intégrations pour RAG, reclasser les documents et modérer le contenu. Chaque API a un point de terminaison et un format différents. + -**Comment OmniRoute le résout :** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Embeddings**— `/v1/embeddings` avec 6 fournisseurs et plus de 9 modèles --**Génération d'images**— `/v1/images/generations` avec 10 fournisseurs et plus de 20 modèles (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Text-to-Video**— `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) et SD WebUI --**Text-to-Music**— `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) --**Transcription audio**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Text-to-Speech**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + fournisseurs existants --**Modérations**— `/v1/moderations` — Contrôles de sécurité du contenu --**Reclassement**— `/v1/rerank` — Reclassement de la pertinence du document --**API Responses**— Prise en charge complète de `/v1/responses` pour le Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. - -🧪 14. "Je n'ai aucun moyen de tester et de comparer la qualité des différents modèles" +**How OmniRoute solves it:** -Les développeurs veulent savoir quel modèle convient le mieux à leur cas d'utilisation (code, traduction, raisonnement) mais la comparaison manuelle est lente. Il n’existe aucun outil d’évaluation intégré. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**Comment OmniRoute le résout :** + --**Évaluations LLM**— Tests Golden Set avec 10 cas préchargés couvrant les salutations, les mathématiques, la géographie, la génération de code, la conformité JSON, la traduction, la démarque, le refus de sécurité --**4 stratégies de correspondance**— `exact`, `contains`, `regex`, `custom` (fonction JS) --**Banc de test Translator Playground**— Tests par lots avec plusieurs entrées et sorties attendues, comparaison entre fournisseurs --**Chat Tester**— Aller-retour complet avec rendu de réponse visuelle --**Live Monitor**— Flux en temps réel de toutes les requêtes transitant par le proxy +
+🌍 12. "The interface is English-only and my team doesn't speak English" - -📈 15. "J'ai besoin d'évoluer sans perdre en performances" +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -À mesure que le volume de demandes augmente, sans mettre en cache les mêmes questions, cela génère des coûts en double. Sans idempotence, les demandes en double gaspillent le traitement. Les limites tarifaires par fournisseur doivent être respectées. +**How OmniRoute solves it:** -**Comment OmniRoute le résout :** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Cache sémantique**— Le cache à deux niveaux (signature + sémantique) réduit les coûts et la latence --**Request Idempotency**— Fenêtre de déduplication de 5 s pour des requêtes identiques --**Détection de limite de débit**— RPM par fournisseur, écart minimum et suivi simultané maximum --**Limites de débit modifiables**— Valeurs par défaut configurables dans Paramètres → Résilience avec persistance --**Cache de validation de clé API**— Cache à 3 niveaux pour les performances de production --**Tableau de bord de santé avec télémétrie**— latence p50/p95/p99, statistiques de cache, disponibilité
+ - -🤖 16. "Je veux contrôler le comportement du modèle à l'échelle mondiale" +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Les développeurs qui souhaitent que toutes les réponses soient dans une langue spécifique, avec un ton spécifique, ou qui souhaitent limiter les jetons de raisonnement. Configurer cela dans chaque outil/demande n’est pas pratique. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Comment OmniRoute le résout :** +**How OmniRoute solves it:** --**Injection d'invite système**— Invite globale appliquée à toutes les requêtes --**Thinking Budget Validation**— Contrôle d'allocation de jetons de raisonnement par requête (passthrough, automatique, personnalisé, adaptatif) --**9 stratégies de routage** – Stratégies globales qui déterminent la manière dont les demandes sont distribuées --**Wildcard Router**— Les modèles `provider/*` acheminent dynamiquement vers n'importe quel fournisseur --**Combo Enable/Disable Toggle**— Basculez les combos directement depuis le tableau de bord --**Provider Toggle**— Activer/désactiver toutes les connexions pour un fournisseur en un seul clic --**Fournisseurs bloqués**— Exclure des fournisseurs spécifiques de la liste `/v1/models`
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex - -🧰 17. "J'ai besoin d'outils MCP en tant que fonctionnalités de produit de premier ordre" + -De nombreuses passerelles IA exposent MCP uniquement en tant que détail d'implémentation caché. Les équipes ont besoin d’une couche opérationnelle visible et gérable. +
+🧪 14. "I have no way to test and compare quality across models" -**Comment OmniRoute le résout :** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP apparaît dans l'onglet de navigation du tableau de bord et de protocole de point de terminaison -- Page de gestion MCP dédiée avec processus, outils, portées et audit -- Démarrage rapide intégré pour `omniroute --mcp` et l'intégration des clients
+**How OmniRoute solves it:** - -🧠 18. "J'ai besoin d'une orchestration A2A avec des chemins de tâches de synchronisation et de flux" +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Les flux de travail des agents nécessitent à la fois des réponses directes et une exécution en continu de longue durée avec contrôle du cycle de vie. + -**Comment OmniRoute le résout :** +
+📈 15. "I need to scale without losing performance" -- Point de terminaison A2A JSON-RPC (`POST /a2a`) avec `message/send` et `message/stream` -- Streaming SSE avec propagation de l'état terminal -- API de cycle de vie des tâches pour "tasks/get" et "tasks/cancel"
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. - -🛰️ 19. "J'ai besoin d'un véritable état de santé du processus MCP, et non d'un état deviné" +**How OmniRoute solves it:** -Les équipes opérationnelles doivent savoir si MCP est réellement actif, et pas seulement si une API est accessible. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**Comment OmniRoute le résout :** + -- Fichier de battement de cœur d'exécution avec PID, horodatages, transport, nombre d'outils et mode de portée -- API de statut MCP combinant battement de coeur + activité récente -- Cartes d'état de l'interface utilisateur pour la fraîcheur des processus/disponibilité/battement de cœur +
+🤖 16. "I want to control model behavior globally" - -📋 20. "J'ai besoin d'une exécution vérifiable de l'outil MCP" +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Lorsque les outils modifient la configuration ou déclenchent des actions opérationnelles, les équipes ont besoin d'une traçabilité médico-légale. +**How OmniRoute solves it:** -**Comment OmniRoute le résout :** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- Journalisation d'audit basée sur SQLite pour les appels d'outils MCP -- Filtres par outil, succès/échec, clé API et pagination -- Tableau d'audit du tableau de bord + points de terminaison de statistiques pour l'automatisation
+ - -🔐 21. "J'ai besoin d'autorisations MCP limitées par intégration" +
+🧰 17. "I need MCP tools as first-class product capabilities" -Différents clients doivent avoir le moindre privilège d’accès aux catégories d’outils. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Comment OmniRoute le résout :** +**How OmniRoute solves it:** -- 10 étendues MCP granulaires pour un accès contrôlé aux outils -- Application de la portée et visibilité dans l'interface utilisateur de gestion MCP -- Posture par défaut sécurisée pour les outils opérationnels
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding - -⚙️ 22. "J'ai besoin de contrôles opérationnels sans redéploiement" + -Les équipes ont besoin de changements d'exécution rapides lors d'incidents ou d'événements de coûts. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**Comment OmniRoute le résout :** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Activer le combo de commutation directement depuis le tableau de bord MCP -- Appliquer des profils de résilience à partir de packs de politiques prédéfinis -- Réinitialiser l'état du disjoncteur à partir du même panneau de commande
+**How OmniRoute solves it:** - -🔄 23. "J'ai besoin d'une visibilité et d'une annulation en direct du cycle de vie des tâches A2A" +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Sans visibilité sur le cycle de vie, les incidents de tâches deviennent difficiles à trier. + -**Comment OmniRoute le résout :** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Liste des tâches/filtrage par état/compétence avec pagination -- Analyse approfondie des métadonnées, des événements et des artefacts des tâches -- Point de terminaison d'annulation de tâche et action de l'interface utilisateur avec confirmation
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. - -🌊 24. "J'ai besoin de métriques de flux actif pour la charge A2A" +**How OmniRoute solves it:** -Les flux de travail de streaming nécessitent une vision opérationnelle de la concurrence et des connexions en direct. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**Comment OmniRoute le résout :** + -- Compteurs de flux actifs intégrés au statut A2A -- Horodatage de la dernière tâche et nombre par état -- Cartes de tableau de bord A2A pour la surveillance des opérations en temps réel +
+📋 20. "I need auditable MCP tool execution" - -🪪 25. "J'ai besoin d'une découverte d'agent standard pour les clients" +When tools mutate config or trigger ops actions, teams need forensic traceability. -Les clients et orchestrateurs externes ont besoin de métadonnées lisibles par machine pour l'intégration. +**How OmniRoute solves it:** -**Comment OmniRoute le résout :** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- Carte d'agent exposée dans `/.well-known/agent.json` -- Capacités et compétences affichées dans l'interface utilisateur de gestion -- L'API de statut A2A inclut des métadonnées de découverte pour l'automatisation
+ - -🧭 26. "J'ai besoin de la possibilité de découvrir le protocole dans l'UX du produit" +
+🔐 21. "I need scoped MCP permissions per integration" -Si les utilisateurs ne peuvent pas découvrir les surfaces de protocole, l’adoption et la qualité du support chutent. +Different clients should have least-privilege access to tool categories. -**Comment OmniRoute le résout :** +**How OmniRoute solves it:** -- Page**Points de terminaison**consolidée avec des onglets pour les points de terminaison Proxy, MCP, A2A et API -- Basculement de l'état du service en ligne (en ligne/hors ligne) pour MCP et A2A -- Liens depuis l'aperçu vers les onglets de gestion dédiés
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling - -🧪 27. "J'ai besoin d'une validation de protocole de bout en bout avec de vrais clients" + -Les tests simulés ne suffisent pas pour valider la compatibilité des protocoles avant la publication. +
+⚙️ 22. "I need operational controls without redeploying" -**Comment OmniRoute le résout :** +Teams need quick runtime changes during incidents or cost events. -- Suite E2E qui démarre l'application et utilise un véritable transport client MCP SDK -- Tests client A2A pour les flux de découverte, d'envoi, de streaming, d'obtention et d'annulation -- Vérifier les assertions par rapport aux API d'audit MCP et de tâches A2A
+**How OmniRoute solves it:** - -📡 28. "J'ai besoin d'une observabilité unifiée sur toutes les interfaces" +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -Le fractionnement de l'observabilité par protocole crée des angles morts et un MTTR plus long. + -**Comment OmniRoute le résout :** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Tableaux de bord/journaux/analyses unifiés dans un seul produit -- Santé + audit + télémétrie des demandes sur les couches OpenAI, MCP et A2A -- API opérationnelles pour le statut et l'automatisation
+Without lifecycle visibility, task incidents become hard to triage. - -💼 29. "J'ai besoin d'un environnement d'exécution pour l'orchestration proxy + outils + agent" +**How OmniRoute solves it:** -L’exécution de nombreux services distincts augmente les coûts opérationnels et les modes de défaillance. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**Comment OmniRoute le résout :** + -- Proxy compatible OpenAI, serveur MCP et serveur A2A dans une seule pile -- Authentification partagée, résilience, stockage de données et observabilité -- Modèle de politique cohérent sur toutes les surfaces d'interaction +
+🌊 24. "I need active stream metrics for A2A load" - -🚀 30. "J'ai besoin d'expédier des flux de travail agentiques sans prolifération de code collant" +Streaming workflows require operational insight into concurrency and live connections. -Les équipes perdent de la vitesse lors de l’assemblage de plusieurs services et scripts ad hoc. +**How OmniRoute solves it:** -**Comment OmniRoute le résout :** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Stratégie de point de terminaison unifiée pour les clients et les agents -- Interfaces utilisateur de gestion de protocole intégrées et chemins de validation de fumée -- Bases prêtes pour la production (sécurité, journalisation, résilience, sauvegarde)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Playbook A : Maximisez l'abonnement payant + sauvegarde bon marché**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -608,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Playbook B : pile de codage à coût nul**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Playbook C : chaîne de secours toujours active 24h/24 et 7j/7**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -631,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Playbook D : Opérations d'agent avec MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Configurez le codage IA en quelques minutes à**0 $/mois**. Connectez ces comptes gratuits et utilisez le combo**Free Stack**intégré. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Étape | Actions | Fournisseurs débloqués | +| Step | Action | Providers Unlocked | | ---- | -------------------------------------------------- | ------------------------------------------------------------------ | -| 1 | Connectez**Kiro**(ID AWS Builder OAuth) | Claude Sonnet 4.5, Haïku 4.5 —**illimité**| -| 2 | Connectez**Qoder**(Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... —**illimité**| -| 3 | Connectez**Qwen**(code de l'appareil) | qwen3-coder-plus, qwen3-coder-flash... —**illimité**| -| 4 | Connectez**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180 000/mois gratuits**| -| 5 | `/dashboard/combos` → Modèle**Free Stack ($0)**| Faites un tourniquet automatique entre tous les fournisseurs gratuits | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Pointez n'importe quel IDE/CLI vers :**`http://localhost:20128/v1` · Clé API : `any-string` · Terminé. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Couverture supplémentaire facultative (également gratuite) :**Clé API Groq (30 RPM gratuits), NVIDIA NIM (40 RPM gratuits, plus de 70 modèles), Cerebras (1 million de tok/jour), clé API LongCat (50 millions de jetons/jour !), Cloudflare Workers AI (10 000 neurones/jour, plus de 50 modèles).## Démarrage Rapide +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Démarrage Rapide ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **Utilisateurs pnpm :**Exécutez `pnpm approved-builds -g` après l'installation pour activer les scripts de build natifs requis par `better-sqlite3` et `@swc/core` : +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > > ```bash > pnpm install -g omniroute -> pnpm approuver-builds -g # Sélectionner tous les packages → approuver +> pnpm approve-builds -g # Select all packages → approve > omniroute > ``` -Le tableau de bord s'ouvre sur « http://localhost:20128 » et l'URL de base de l'API est « http://localhost:20128/v1 ». +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Commande | Descriptif | -| ----------------------- | --------------------------------------------------------------------------- | -| `omniroute` | Démarrer le serveur (`PORT=20128`, API et tableau de bord sur le même port) | -| `omniroute --port 3000` | Définir le port canonique/API sur 3000 | -| `omniroute --mcp` | Démarrer le serveur MCP (transport stdio) | -| `omniroute --no-open` | Ne pas ouvrir automatiquement le navigateur | -| `omniroute --help` | Afficher l'aide | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Mode de port partagé en option :```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -Pour la plupart des déploiements, vous n'avez besoin que de : +For most deployments, you only need: -| Variables | Par défaut | Objectif | -| -------------------- | ----------------------------- | -------------------------------------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | «600 000» | Base de référence partagée pour la récupération en amont, les délais d'attente Undici cachés, les demandes d'empreintes digitales TLS et les délais d'attente des demandes de pont API/proxy | -| `STREAM_IDLE_TIMEOUT_MS` | hérite de `REQUEST_TIMEOUT_MS` | Écart maximum entre les morceaux de streaming avant qu'OmniRoute n'abandonne le flux SSE | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -La compatibilité ascendante est préservée : `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` et d'autres variables de délai d'expiration par couche fonctionnent toujours et remplacent la ligne de base partagée. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Des remplacements avancés sont disponibles si vous avez besoin d'un contrôle plus précis :| Variables | Par défaut | Objectif | -| --------------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | hérite de `REQUEST_TIMEOUT_MS` | Délai d'expiration total de la demande en amont utilisé par le signal d'abandon de récupération principal | -| `FETCH_HEADERS_TIMEOUT_MS` | hérite de `FETCH_TIMEOUT_MS` | Délai Undici pour la réception des en-têtes de réponse en amont | -| `FETCH_BODY_TIMEOUT_MS` | hérite de `FETCH_TIMEOUT_MS` | Limite de temps Undici entre les morceaux de corps en amont (`0` le désactive) | -| `FETCH_CONNECT_TIMEOUT_MS` | '30 000' | Undici Délai d'expiration de la connexion TCP | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | '4000' | Undici délai d'attente du socket keep-alive inactif | -| `TLS_CLIENT_TIMEOUT_MS` | hérite de `FETCH_TIMEOUT_MS` | Délai d'expiration pour les demandes d'empreintes digitales TLS effectuées via `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | hérite de `REQUEST_TIMEOUT_MS` ou `30000` | Délai d'expiration pour le transfert du proxy `/v1` du port API vers le port du tableau de bord | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300 000)` | Délai d'expiration des requêtes entrantes sur le serveur de pont API | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | «60 000» | Délai d'expiration de l'en-tête entrant sur le serveur de pont API | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | '5000' | Délai d'expiration de conservation sur le serveur de pont API | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Délai d'inactivité du socket sur le serveur de pont API (`0` le désactive) | +Advanced overrides are available if you need finer control: -Si vous exécutez OmniRoute derrière Nginx, Caddy, Cloudflare ou un autre proxy inverse, assurez-vous que le proxy -les délais d'attente sont également supérieurs aux délais d'attente de votre flux/récupération OmniRoute.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Ouvrez le tableau de bord → « Fournisseurs » et connectez au moins un fournisseur (OAuth ou clé API). -2. Ouvrez le tableau de bord → « Endpoints » et créez une clé API. -3. (Facultatif) Ouvrez le tableau de bord → « Combos » et définissez votre chaîne de secours.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Fonctionne avec les SDK compatibles Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode et OpenAI.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (pour les opérations pilotées par outils) :**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -Connectez ensuite votre client MCP via `stdio` et testez des outils tels que : +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (pour les flux de travail d'agent à agent) :**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -760,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Cette suite valide les flux clients MCP et A2A réels par rapport à une application en cours d'exécution.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -768,14 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` - +
+Void Linux (`xbps-src` template) -Annuler Linux (modèle `xbps-src`) - -Pour les utilisateurs de Void Linux, vous pouvez créer un package natif en utilisant `xbps-src`. Enregistrez ce bloc sous `srcpkgs/omniroute/template` :```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -787,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -795,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -871,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -882,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute est disponible sous forme d'image Docker publique sur [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Exécution rapide :**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -892,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**Avec fichier d'environnement :**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Utilisation de Docker Compose :**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -La prise en charge des tableaux de bord pour les déploiements Docker inclut désormais un**Cloudflare Quick Tunnel**en un clic sur « Tableau de bord → Points de terminaison ». La première activation télécharge « cloudflared » uniquement en cas de besoin, démarre un tunnel temporaire vers votre point de terminaison « /v1 » actuel et affiche l'URL « https://\*.trycloudflare.com/v1 » générée directement sous votre URL publique normale. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Remarques : +Notes: -- Les URL du tunnel rapide sont temporaires et changent après chaque redémarrage. -- Les tunnels rapides ne sont pas automatiquement restaurés après un redémarrage d'OmniRoute ou d'un conteneur. Réactivez-les depuis le tableau de bord si nécessaire. -- L'installation gérée prend actuellement en charge Linux, macOS et Windows sur `x64` / `arm64`. -- Les tunnels rapides gérés utilisent par défaut le transport HTTP/2 pour éviter les avertissements de tampon QUIC UDP bruyants dans les environnements de conteneurs contraints. Définissez `CLOUDFLARED_PROTOCOL=quic` ou `auto` si vous souhaitez un transport différent. -- Les images Docker regroupent les racines de l'autorité de certification du système et les transmettent au « cloudflared » géré, ce qui évite les échecs de confiance TLS lorsque le tunnel s'amorce à l'intérieur du conteneur. -- SQLite fonctionne en mode WAL. `docker stop` doit être autorisé à se terminer afin qu'OmniRoute puisse vérifier les dernières modifications dans `storage.sqlite`. -- Les fichiers Compose fournis définissent déjà un délai de grâce d'arrêt de 40 s. Si vous exécutez l'image directement, conservez « --stop-timeout 40 » (ou similaire) afin que les arrêts manuels n'interrompent pas le nettoyage de l'arrêt. -- Définissez `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` si vous souhaitez qu'OmniRoute utilise un binaire existant au lieu d'en télécharger un. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Utilisation de Docker Compose avec Caddy (HTTPS Auto-TLS) :** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute peut être exposé en toute sécurité grâce au provisionnement SSL automatique de Caddy. Assurez-vous que l'enregistrement DNS A de votre domaine pointe vers l'adresse IP de votre serveur.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` +| Image | Tag | Size | Description | +| ------------------------ | -------- | ------ | --------------------- | +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | -| Images | Étiquette | Taille | Descriptif | -| -------------------- | -------- | ------ | ------------------------------------ | -| `diegosouzapw/omniroute` | `dernier` | ~250 Mo | Dernière version stable | -| `diegosouzapw/omniroute` | `1.0.3` | ~250 Mo | Version actuelle |--- +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**NOUVEAU !**OmniRoute est désormais disponible en tant qu'**application de bureau native**pour Windows, macOS et Linux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Exécutez OmniRoute en tant qu'application de bureau autonome : aucun terminal, aucun navigateur, aucune connexion Internet requise pour les modèles locaux. L'application basée sur Electron comprend : +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Fenêtre native**— Fenêtre d'application dédiée avec intégration dans la barre d'état système -- 🔄**Démarrage automatique**— Lancez OmniRoute lors de la connexion au système -- 🔔**Notifications natives**— Recevez des alertes en cas d'épuisement de quota ou de problèmes de fournisseur -- ⚡**Installation en un clic**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Mode hors ligne**— Fonctionne entièrement hors ligne avec le serveur fourni### Démarrage Rapide +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Démarrage Rapide ```bash # Development mode @@ -981,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -Lorsqu'il est réduit, OmniRoute réside dans votre barre d'état système avec des actions rapides : +When minimized, OmniRoute lives in your system tray with quick actions: -- Ouvrir le tableau de bord -- Changer le port du serveur -- Quitter l'application +- Open dashboard +- Change server port +- Quit application -📖 Documentation complète : [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Niveau | Fournisseur | Coût | Réinitialisation des quotas | Idéal pour | -| ----------------- | --------------------------------- | ---------------------------------------- | --------------------------- | ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 ABONNEMENT** | Claude Code (Pro) | 20 $/mois | 5h + hebdomadaire | Déjà abonné | -| | Codex (Plus/Pro) | 20-200 $/mois | 5h + hebdomadaire | Utilisateurs d'OpenAI | -| | CLI Gémeaux | **GRATUIT** | 180K/mois + 1K/jour | Tout le monde! | -| | Copilote GitHub | 10-19 $/mois | Mensuel | Utilisateurs GitHub | -| **🔑 CLÉ API** | NIM NVIDIA | **GRATUIT**(développement pour toujours) | ~40 tr/min | Plus de 70 modèles ouverts | -| | Cérébraux | **GRATUIT**(1 million de tok/jour) | 60 000 TPM / 30 tr/min | Le plus rapide du monde | -| | Groq | **GRATUIT**(30 TR/MIN) | 14,4 000 tr/min | Lama/Gemma ultra-rapide | -| | DeepSeek V3.2 | 0,27 $/1,10 $ par 1 million | Aucun | Raisonnement meilleur prix/qualité | -| | xAI Grok-4 Rapide | **0,20$/0,50$ par 1M**🆕 | Aucun | Appel d'outil le plus rapide +, ultra lent | -| | xAI Grok-4 (standard) | 0,20 $/1,50 $ par 1 M 🆕 | Aucun | Produit phare du raisonnement de xAI | -| | Mistral | Essai gratuit + payant | Tarif limité | IA européenne | -| | OuvrirRouter | Paiement à l'utilisation | Aucun | Plus de 100 modèles agrégés. | -| **💰 BON MARCHÉ** | GLM-5 (via Z.AI) 🆕 | 0,5 $/1 M | Tous les jours 10h | Sortie 128K, nouveau produit phare | -| | GLM-4.7 | 0,6 $/1 M | Tous les jours 10h | Sauvegarde budgétaire | -| | MiniMax M2.5 🆕 | Entrée de 0,3 $/1 M | 5 heures roulantes | Raisonnement + tâches agentiques | -| | MiniMax M2.1 | 0,2 $/1 M | 5 heures roulantes | Option la moins chère | -| | Kimi K2.5 (API Moonshot) 🆕 | Paiement à l'utilisation | Aucun | Accès direct à l'API Moonshot | -| | Kimi K2 | 9 $/mois plat | 10 millions de jetons/mois | Coût prévisible | -| **🆓 GRATUIT** | Qoder | **0$** | Unlimited | 5 modèles illimités | -| | Qwen | **0$** | Illimité | 4 modèles illimités | -| | Kiro | **0$** | Illimité | Claude Sonnet/Haïku (AWS Builder) | -| | LongCat Flash-Lite 🆕 | **$0**(50M tok/jour 🔥) | 1 RPS | Le plus grand quota gratuit sur Terre | -| | Pollinisations IA 🆕 | **0$**(aucune clé requise) | 1 demande/15s | GPT-5, Claude, DeepSeek, Lama 4 | -| | IA des travailleurs Cloudflare 🆕 | **0$**(10 000 neurones/jour) | ~150 resp/jour | Plus de 50 modèles, avantage mondial | -| | IA Scaleway 🆕 | **0 $**(1 million de jetons au total) | Tarif limité | UE/RGPD, Qwen3 235B, Lama 70B | > 🆕**Nouveaux modèles ajoutés (mars 2026) :**Famille Grok-4 Fast à 0,20 $/0,50 $/M (référence à 1 143 ms – 30 % plus rapide que Gemini 2.5 Flash), GLM-5 via Z.AI avec sortie 128K, raisonnement MiniMax M2.5, tarification mise à jour DeepSeek V3.2, Kimi K2.5 via l'API directe Moonshot. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 Pile combinée à 0 $ — La configuration gratuite complète :**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Zéro coût. N'arrête jamais de coder.**Configurez-le comme un combo OmniRoute et toutes les solutions de secours se produisent automatiquement — sans jamais de commutation manuelle.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Tous les modèles ci-dessous sont**100 % gratuits et aucune carte de crédit requise**. OmniRoute effectue un acheminement automatique entre eux lorsqu'un quota est épuisé : combinez-les tous pour un combo incassable à 0 $.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Modèle | Préfixe | Limite | Limite de taux | -| ------------------- | ------ | ------------- | ------------------------------------ | -| `claude-sonnet-4.5` | `kr/` |**Illimité**| Aucun plafond quotidien signalé | -| `claude-haïku-4.5` | `kr/` |**Illimité**| Aucun plafond quotidien signalé | -| `claude-opus-4.6` | `kr/` |**Illimité**| Dernier Opus via Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) -| Modèle | Préfixe | Limite | Limite de taux | +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | --------------------- | +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | + +### 🟢 QODER MODELS (Free PAT via qodercli) + +| Model | Prefix | Limit | Rate Limit | | ------------------ | ------ | ------------- | --------------- | -| `kimi-k2-pensée` | `si/` |**Illimité**| Aucun plafond signalé | -| `qwen3-coder-plus` | `si/` |**Illimité**| Aucun plafond signalé | -| `deepseek-r1` | `si/` |**Illimité**| Aucun plafond signalé | -| `minimax-m2.1` | `si/` |**Illimité**| Aucun plafond signalé | -| `kimi-k2` | `si/` |**Illimité**| Aucun plafond signalé | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -> Méthode de connexion recommandée :**Jeton d'accès personnel + `qodercli`**. Le navigateur OAuth est -> expérimental et désactivé par défaut sauf si les variables d'environnement `QODER_OAUTH_*` sont configurées.### 🟡 QWEN MODELS (Device Code Auth) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| Modèle | Préfixe | Limite | Limite de taux | +### 🟡 QWEN MODELS (Device Code Auth) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | ------------------- | -| `qwen3-coder-plus` | `qw/` |**Illimité**| Aucun plafond signalé | -| `qwen3-coder-flash` | `qw/` |**Illimité**| Aucun plafond signalé | -| `qwen3-coder-suivant` | `qw/` |**Illimité**| Aucun plafond signalé | -| `modèle-vision` | `qw/` |**Illimité**| Multimodal (images) |### 🟣 GEMINI CLI (Google OAuth) +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Modèle | Préfixe | Limite | Limite de taux | -| -------------------- | ------ | -------------------------------- | ------------- | -| `gemini-3-flash-aperçu` | `gc/` |**180K tok/mois**+ 1K/jour | Réinitialisation mensuelle | -| `gemini-2.5-pro` | `gc/` | 180K/mois (piscine partagée) | Haute qualité |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +### 🟣 GEMINI CLI (Google OAuth) -| Niveau | Limite quotidienne | Limite de taux | Remarques | +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | + +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---------- | ------------ | ----------- | ------------------------------------------------------ | -| Gratuit (développement) | Pas de plafond de jetons |**~40 tr/min**| Plus de 70 modèles ; transition vers des limites de taux pures mi-2025 | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -Modèles gratuits populaires : `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -| Niveau | Limite quotidienne | Limite de taux | Remarques | -| ---- | ----------------- | ---------------- | ------------------------------------------------ | -| Gratuit |**1 million de jetons/jour**| 60 000 TPM / 30 tr/min | L'inférence LLM la plus rapide au monde ; resets daily | +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -Disponible gratuitement : `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -| Niveau | Limite quotidienne | Limite de taux | Remarques | +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` + +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---- | ------------- | ---------------- | ----------------------------------------- | -| Gratuit |**14 400 RPJ**| 30 tr/min par modèle | Pas de carte de crédit ; 429 en limite, non facturé | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -Disponible gratuitement : `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -| Modèle | Préfixe | Quota quotidien gratuit | Remarques | +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 + +| Model | Prefix | Daily Free Quota | Notes | | ----------------------------- | ------ | ----------------- | ----------------------- | -| `LongCat-Flash-Lite` | `lc/` |**50 millions de jetons**💥 | Le plus grand quota gratuit jamais vu | -| `LongCat-Flash-Chat` | `lc/` | 500 000 jetons | Chat multi-tours | -| `LongCat-Flash-Pensée` | `lc/` | 500 000 jetons | Raisonnement / CoT | -| `LongCat-Flash-Pensée-2601` | `lc/` | 500 000 jetons | Version janvier 2026 | -| `LongCat-Flash-Omni-2603` | `lc/` | 500 000 jetons | Multimodal | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | -> 100 % gratuit en version bêta publique. Inscrivez-vous sur [longcat.chat](https://longcat.chat) par e-mail ou par téléphone. Se réinitialise quotidiennement à 00h00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. -| Modèle | Préfixe | Limite de taux | Fournisseur derrière | +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| `openai` | `pol/` | 1 demande/15s | GPT-5 | -| `claude` | `pol/` | 1 demande/15s | Claude Anthropique | -| `Gémeaux` | `pol/` | 1 demande/15s | Google Gémeaux | -| `recherche profonde` | `pol/` | 1 demande/15s | Recherche profonde V3 | -| `lama` | `pol/` | 1 demande/15s | Meta Lama 4 Scout | -| 'mistral' | `pol/` | 1 demande/15s | Mistral IA | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**Zéro friction :**Pas d'inscription, pas de clé API. Ajoutez le fournisseur Pollinations avec un champ clé vide et cela fonctionne immédiatement.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| Niveau | Neurones quotidiens | Utilisation équivalente | Remarques | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 + +| Tier | Daily Neurons | Equivalent Usage | Notes | | ---- | ------------- | --------------------------------------- | ----------------------- | -| Gratuit |**10 000**| ~ 150 LLM resp / 500 s audio / 15 000 intégrations | Avantage mondial, plus de 50 modèles | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -Modèles gratuits populaires : `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (audio gratuit !), `@cf/qwen/qwen2.5-coder-15b-instruct` +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -> Nécessite un jeton API + un identifiant de compte de [dash.cloudflare.com](https://dash.cloudflare.com). Stockez l’ID de compte dans les paramètres du fournisseur.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -| Niveau | Quotas gratuits | Localisation | Remarques | +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | | ---- | ------------- | ------------ | ----------------------------------- | -| Gratuit |**1 million de jetons**| 🇫🇷 Paris, UE | Aucune carte de crédit nécessaire dans certaines limites | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | -Disponible gratuitement : `qwen3-235b-a22b-instruct-2507` (Qwen3 235B !), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` -> Conforme UE/RGPD. Obtenez la clé API sur [console.scaleway.com](https://console.scaleway.com). +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). ->**💡 The Ultimate Free Stack (11 fournisseurs, 0 $ pour toujours) :** +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Sonnet/Haiku ILLIMITÉ -> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 ILLIMITÉ -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 millions de jetons/jour 🔥 -> Pollinisations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — aucune clé nécessaire -> Qwen (qw/) → modèles qwen3-coder ILLIMITÉS -> Gemini (gemini/) → Gemini 2.5 Flash — 1 500 req/jour gratuits -> Cloudflare AI (cf/) → 50+ modèles — 10K Neurons/jour -> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1 million de jetons gratuits (UE) -> Groq (groq/) → Lama/Gemma — 14,4K req/jour ultra-rapide -> NVIDIA NIM (nvidia/) → Plus de 70 modèles ouverts — 40 RPM pour toujours -> Cerebras (cerebras/) → Lama/Qwen le plus rapide au monde — 1 million de tok/jour -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Transcrivez n'importe quel audio/vidéo pour**0 $**— Deepgram mène avec 200 $ gratuits, AssemblyAI 50 $ de secours, Groq Whisper comme sauvegarde d'urgence illimitée. +## 🎙️ Free Transcription Combo -| Fournisseur | Crédits gratuits | Meilleur modèle | Limite de taux | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. + +| Provider | Free Credits | Best Model | Rate Limit | | ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | -| 🟢**Deepgram**|**200 $ gratuits**(inscription) | `nova-3` — meilleure précision, plus de 30 langues | Aucune limite de RPM sur les crédits gratuits | -| 🔵**AssemblyAI**|**50 $ gratuits**(inscription) | `universal-3-pro` — chapitres, sentiments, PII | Aucune limite de RPM sur les crédits gratuits | -| 🔴**Groq**|**Gratuit pour toujours**| `chuchotement-large-v3` — OpenAI Whisper | 30 tr/min (taux limité) | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | -**Combo suggéré dans `/dashboard/combos` :**``` +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Ensuite, dans `/dashboard/media` → onglet**Transcription**: téléchargez n'importe quel fichier audio ou vidéo → sélectionnez votre point de terminaison combo → obtenez la transcription dans les formats pris en charge.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 est conçu comme une plate-forme opérationnelle et non comme un simple proxy relais.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Fonctionnalité | Ce qu'il fait | -| --------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Famille rapide Grok-4** | Modèles xAI à 0,20 $/0,50 $/M — 1 143 ms de référence (30 % plus rapide que Gemini 2.5 Flash) | -| 🧠**GLM-5 via Z.AI** | Contexte de sortie de 128 000 $, 0,5 / 1 million de dollars – le dernier produit phare de la famille GLM | -| 🔮**MiniMax M2.5** | Raisonnement + tâches agentiques à 0,30 $/1 million — mise à niveau significative depuis M2.1 | -| 🎯**toolCalling Flag par modèle** | `toolCalling : true/false` par modèle dans le registre — AutoCombo ignore les modèles non compatibles avec les outils | -| 🌍**Détection d'intention multilingue** | Mots-clés PT/ZH/ES/AR dans la notation AutoCombo — meilleure sélection de modèles pour le contenu non anglais | -| 📊**Replis basés sur des benchmarks** | Latence p95 réelle à partir de la notation combinée des flux de requêtes en direct — AutoCombo apprend à partir des données réelles | -| 🔁**Demander une déduplication** | Fenêtre de déduplication basée sur le hachage de contenu — sécurisée multi-agents, évite les frais en double | -| 🔌**Stratégie de routeur enfichable** | Interface extensible `RouterStrategy` — ajoutez une logique de routage personnalisée sous forme de plugins | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Fonctionnalité | Ce qu'il fait | -| -------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Aire de jeux modèle** | Page de tableau de bord pour tester n'importe quel modèle directement — sélecteurs de fournisseur/modèle/point de terminaison, éditeur de Monaco, streaming, abandon, timing | -| 🔏**Correspondance d'empreintes digitales CLI** | Ordre des en-têtes/corps par fournisseur pour correspondre aux signatures CLI natives : basculez par fournisseur dans Paramètres > Sécurité.**Votre IP proxy est préservée** | -| 🤝**Prise en charge ACP (Protocole Agent Client)** | Découverte d'agent CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 de plus), générateur de processus, point de terminaison `/api/acp/agents` | -| 🤖**Tableau de bord des agents ACP** | Page Débogage > Agents — grille de 14 agents avec état d'installation, version, formulaire d'agent personnalisé pour n'importe quel outil CLI. Les utilisateurs**OpenCode**bénéficient d'un bouton "Télécharger opencode.json" qui génère automatiquement une configuration prête à l'emploi avec tous les modèles disponibles. | -| 🔧**Modèle personnalisé de routage `apiFormat`** | Les modèles personnalisés avec `apiFormat : "responses"` sont désormais correctement acheminés vers le traducteur de l'API Responses | -| 🏢**Isolement de l'espace de travail Codex** | Plusieurs espaces de travail Codex par e-mail — OAuth sépare correctement les connexions par ID d'espace de travail | -| 🔄**Mise à jour automatique électronique** | L'application de bureau vérifie les mises à jour + installation automatique au redémarrage | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Fonctionnalité | Ce qu'il fait | -| ---------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**Serveur MCP (25 outils)** | Outils IDE/agent via 3 transports : stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 cœurs + 3 mémoires + 4 outils de compétences | -| 🤝**Serveur A2A (JSON-RPC + SSE)** | Exécution de tâches d'agent à agent avec flux de synchronisation et de streaming | -| 🧭**Page des points de terminaison consolidés** | Page de gestion à onglets avec les onglets Endpoint Proxy, MCP, A2A et API Endpoints | -| 🎚️**Bascules d'activation/désactivation du service** | Interrupteurs ON/OFF pour MCP et A2A avec persistance des paramètres (par défaut : OFF) | -| 🛰️**Battement de coeur d'exécution MCP** | Statut réel du processus (pid, disponibilité, âge du battement de cœur, transport, mode scope) | -| 📋**Piste d'audit MCP** | Journaux d'audit filtrables avec succès/échec et attribution des clés | -| 🔐**Application du champ d'application du MCP** | 10 autorisations de portée granulaire pour un accès contrôlé aux outils | -| 📡**Gestion du cycle de vie des tâches A2A** | Répertorier/filtrer les tâches, inspecter les événements/artefacts, annuler les tâches en cours | -| 📋**Découverte de la carte d'agent** | `/.well-known/agent.json` pour la découverte automatique du client | -| 🧪**Harnais de test du protocole E2E** | Le vrai client MCP SDK + A2A circule dans `test:protocols:e2e` | -| ⚙️**Contrôles opérationnels** | Changer de combo, appliquer des profils de résilience, réinitialiser les disjoncteurs à partir d'une surface de contrôle | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Fonctionnalité | Ce qu'il fait | -| --------------------------------------------- | ----------------------------------------------------------------------------------------------- | ----------------------- | -| 🎯**Repli intelligent à 4 niveaux** | Auto-route : Abonnement → Clé API → Pas cher → Gratuit | -| 📊**Suivi des quotas en temps réel** | Nombre de jetons en direct + réinitialisation du compte à rebours par fournisseur | -| 🔄**Traduction de formats** | OpenAI ↔ Claude ↔ Gémeaux ↔ Réponses avec conversions sécurisées | -| 👥**Support multi-comptes** | Plusieurs comptes par fournisseur avec sélection intelligente | -| 🔄**Actualisation automatique des jetons** | Les jetons OAuth s'actualisent automatiquement avec une nouvelle tentative | -| 🎨**Combos personnalisés** | 9 stratégies d'équilibrage + contrôle de la chaîne de repli | -| 🌐**Routeur générique** | `provider/*` routage dynamique | -| 🧠**Penser les contrôles budgétaires** | Limites du raisonnement passthrough, automatique, personnalisé et adaptatif | -| 🔀**Alias ​​de modèle** | Alias ​​de modèle intégré et personnalisé et sécurité de la migration | -| ⚡**Dégradation de l'arrière-plan** | Acheminer les tâches en arrière-plan de faible priorité vers des modèles moins chers | -| 🧪**Routage intelligent sensible aux tâches** | Modèle de sélection automatique par type de contenu (codage/vision/analyse/résumé) | -| 🔄**Flux de travail des agents A2A** | Orchestrateur FSM déterministe pour les exécutions d'agents multi-étapes avec état | -| 🔀**Routage adaptatif** | Remplacement de stratégie dynamique basé sur le volume de jetons et la complexité des invites | -| 🎲**Diversité des fournisseurs** | Score d'entropie de Shannon équilibrant la distribution du trafic auto-combo | -| 💬**Injection d'invite du système** | Contrôles comportementaux globaux appliqués de manière cohérente | -| 📄**Compatibilité API des réponses** | Prise en charge complète de `/v1/responses` pour le Codex et les flux de travail agents avancés | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Fonctionnalité | Ce qu'il fait | -| ------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**Génération d'images** | `/v1/images/generations` avec le cloud et les backends locaux | -| 📐**Intégrations** | `/v1/embeddings` pour les pipelines de recherche et RAG | -| 🎤**Transcription audio** | `/v1/audio/transcriptions` — 7 fournisseurs (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), détection automatique de langue, prise en charge MP4/MP3/WAV | -| 🔊**Texte-parole** | `/v1/audio/speech` — 10 fournisseurs (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) avec des messages d'erreur corrects | -| 🎬**Génération vidéo** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | -| 🎵**Génération musicale** | `/v1/music/generations` (flux de travail ComfyUI) | -| 🛡️**Modérations** | Contrôles de sécurité `/v1/moderations` | -| 🔀**Reclassement** | `/v1/rerank` pour la notation de pertinence | -| 🔍**Recherche Web**🆕 | `/v1/search` — 5 fournisseurs (Serper, Brave, Perplexity, Exa, Tavily), plus de 6 500 gratuits/mois, basculement automatique, cache | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Fonctionnalité | Ce qu'il fait | -| --------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**Disjoncteurs** | Déclenchement/récupération par modèle avec contrôles de seuil | -| 🎯**Modèles prenant en compte les points de terminaison** | Les modèles personnalisés déclarent les points de terminaison pris en charge + le format API | -| 🛡️**Troupeau anti-tonnerre** | Protections mutex + sémaphore sur les événements de nouvelle tentative/taux | -| 🧠**Cache sémantique + signature** | Réduction des coûts/latences avec deux couches de cache | -| ⚡**Demande d'idempotence** | Fenêtre de protection contre les doubles | -| 🔒**Usurpation d'empreintes digitales TLS** | Empreinte digitale TLS de type navigateur —**réduit la détection des robots et le signalement des comptes** | -| 🔏**Correspondance d'empreintes digitales CLI** | Correspond aux signatures de requêtes CLI natives —**réduit le risque d'interdiction tout en préservant l'adresse IP du proxy** | -| 🌐**Filtrage IP** | Contrôle des listes autorisées/bloquées pour les déploiements exposés | -| 📊**Limites de taux modifiables** | Limites configurables au niveau global/fournisseur avec persistance | -| 📉**Dégradation gracieuse** | Capacités de secours multicouches protégeant les opérations principales de la passerelle | -| 📜**Piste d'audit de configuration** | Suivi des modifications basé sur les différences empêchant la dérive opérationnelle avec de simples restaurations | -| ⏳**Synchronisation de la santé du fournisseur** | Surveillance proactive de l'expiration des jetons déclenchant des alertes avant les échecs d'autorisation | -| 🚪**Désactivation automatique des comptes interdits** | Disjoncteur opérationnel scellant automatiquement les comptes de jetons bloqués de manière permanente | -| 🔑**Gestion des clés API + Cadrage** | Émission/rotation des clés sécurisées et contrôles des modèles/fournisseurs | -| 👁️**Révélation de clé API étendue**🆕 | Récupération opt-in des clés API via `ALLOW_API_KEY_REVEAL` | -| 🛡️**Protégé `/models`** | Gating d'authentification et masquage du fournisseur en option pour le catalogue de modèles | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Fonctionnalité | Ce qu'il fait | -| ------------------------------------------ | ------------------------------------------------------------------------------------ | ---------------------------- | -| 📝**Demande + Journalisation proxy** | Journalisation complète des requêtes/réponses et du proxy | -| 📉**Journaux détaillés diffusés**🆕 | Reconstruit proprement les flux de charge utile SSE dans l'interface utilisateur | -| 📋**Tableau de bord des journaux unifiés** | Vues de requête, de proxy, d'audit et de console sur une seule page | -| 🔍**Demander une télémétrie** | Latence p50/p95/p99 et suivi des requêtes | -| 🏥**Tableau de bord de santé** | Temps de disponibilité, états des disjoncteurs, verrouillages, statistiques du cache | -| 💰**Suivi des coûts** | Contrôles budgétaires et visibilité des prix par modèle | -| 📈**Visualisations analytiques** | Informations sur l'utilisation du modèle/fournisseur et vues des tendances | -| 🧪**Cadre d'évaluation** | Tests du Golden Set avec stratégies de correspondance configurables | -| 📡**Diagnostics en direct**🆕 | Contournement du cache sémantique pour des tests combo précis en direct | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Fonctionnalité | Ce qu'il fait | -| ---------------------------------------- | ----------------------------------------------------------------------------------------- | --------------------- | -| 🌐**Déployer n'importe où** | Localhost, VPS, Docker, environnements Cloud | -| 🚇**Tunnel Cloudflare**🆕 | Intégration Quick Tunnel en un clic depuis le tableau de bord | -| 🔑**Filtrage des modèles de clés API** | Réponse native /v1/models filtrée via les rôles contextuels Bearer attribués | -| ⚡**Contournement intelligent du cache** | Heuristiques TTL configurables et contrôles de récupération forcée | -| 🔄**Sauvegarde/Restauration** | Flux d'exportation/importation et de reprise après sinistre | -| 🧙**Assistant d'intégration** | Configuration guidée de première exécution | -| 🔧**Tableau de bord des outils CLI** | Configuration en un clic pour les outils de codage populaires | -| 🎮**Aire de jeux modèle** | Testez n’importe quel fournisseur/modèle/point de terminaison à partir du tableau de bord | -| 🔏**Bascule d'empreinte digitale CLI** | Correspondance des empreintes digitales par fournisseur dans Paramètres > Sécurité | -| 🌐**i18n (30 langues)** | Tableau de bord complet + prise en charge des langues des documents avec couverture RTL | -| 🧹**Effacer tous les modèles** | Suppression de la liste de modèles en un clic dans les détails du fournisseur | -| 👁️**Contrôles de la barre latérale**🆕 | Masquer les composants et les intégrations dans les paramètres d'apparence | -| 📋**Modèles de problèmes** | Modèles GitHub standardisés pour les bogues et les fonctionnalités | -| 📂**Répertoire de données personnalisé** | Remplacement de `DATA_DIR` pour l'emplacement de stockage | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1294,105 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -En cas d'échec d'un quota, d'un taux ou d'un état de santé, OmniRoute passe automatiquement au candidat suivant sans commutation manuelle.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A sont détectables dans l'interface utilisateur et la documentation (non masquées) -- Les API d'état du protocole exposent les données opérationnelles en direct (`/api/mcp/*`, `/api/a2a/*`) -- Les tableaux de bord incluent des actions pour les opérations du jour 2 (basculements combinés, réinitialisations de disjoncteur, annulation de tâches)#### Translator + validation workflow +#### Protocol management that is visible and operable -La zone Traducteur comprend : +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Playground** : demander des contrôles de transformation -**Testeur de chat** : aller-retour complet de requête/réponse -**Banc de test** : plusieurs cas en une seule fois -**Live Monitor** : vue du trafic en temps réel +#### Translator + validation workflow -Plus validation du protocole avec de vrais clients via `npm run test:protocols:e2e`. +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Référence de l'outil, configurations IDE et exemples de clients +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A Server README](src/lib/a2a/README.md)**— Compétences, méthodes JSON-RPC, streaming et cycle de vie des tâches## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute comprend un cadre d'évaluation intégré pour tester la qualité des réponses LLM par rapport à un ensemble de référence. Accédez-y via**Analytics → Evals**dans le tableau de bord.### Built-in Golden Set +## 🧪 Evaluations (Evals) -Le « OmniRoute Golden Set » préchargé contient des cas de test pour : +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Salutations, mathématiques, géographie, génération de code -- Conformité au format JSON, traduction, génération de démarques -- Refus de sécurité (contenu nuisible), comptage, logique booléenne### Evaluation Strategies +### Built-in Golden Set -| Stratégie | Descriptif | Exemple | -| ---------------------- | --------------------------------------------------------------- | ---------------------------------- | --- | -| `exact` | La sortie doit correspondre exactement | `"4"` | -| `contient` | La sortie doit contenir une sous-chaîne (insensible à la casse) | `"Paris"` | -| `expression régulière` | La sortie doit correspondre au modèle regex | `"1.*2.*3"` | -| `personnalisé` | La fonction JS personnalisée renvoie vrai/faux | `(sortie) => sortie.longueur > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) - +
+🧩 MCP Setup (Model Context Protocol) -🧩 Configuration MCP (Model Context Protocol) +Start MCP transport in stdio mode: -Démarrez le transport MCP en mode stdio :```bash +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Flux de validation recommandé : +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Connectez votre client MCP via stdio. -2. Exécutez `omniroute_get_health`. -3. Exécutez `omniroute_list_combos`. -4. Ouvrez « /dashboard/mcp » pour confirmer le rythme cardiaque, l'activité et l'audit. - -API utiles pour l'automatisation : +Useful APIs for automation: - `GET /api/mcp/status` - `GET /api/mcp/tools` - `GET /api/mcp/audit` -- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats` - -🤝 Configuration A2A (Agent2Agent) + -Découvrez l'agent :```bash +
+🤝 A2A Setup (Agent2Agent) + +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Envoyer une tâche :```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` +Manage lifecycle: -Gérer le cycle de vie : - -- `GET /api/a2a/statut` -- `GET /api/a2a/tâches` +- `GET /api/a2a/status` +- `GET /api/a2a/tasks` - `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -Interface utilisateur opérationnelle : +Operational UI: -- `/dashboard/a2a` pour l'observabilité des tâches/états/flux et les actions de fumée
+- `/dashboard/a2a` for task/state/stream observability and smoke actions - -🧪 Validation du protocole de bout en bout + -Validez les deux protocoles avec de vrais clients :```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Cela vérifie : +This verifies: -- Connexion/liste/appel du client MCP SDK -- Découverte A2A/envoyer/diffuser/obtenir/annuler -- Recoupement des données dans les API d'audit MCP et de gestion des tâches A2A
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs - + -💳 Fournisseurs d'abonnement### Claude Code (Pro/Max) +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1405,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Conseil de pro :**Utilisez Opus pour les tâches complexes, Sonnet pour la rapidité. OmniRoute suit le quota par modèle !### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1419,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Chaque compte Codex dispose désormais de bascules de politique dans « Tableau de bord -> Fournisseurs » : +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (ON/OFF) : applique la politique de seuil de fenêtre de 5 heures. -- `Hebdomadaire` (ON/OFF) : applique la politique de seuil de fenêtre hebdomadaire. -- Comportement de seuil : lorsqu'une fenêtre activée atteint >=90 % d'utilisation, ce compte est ignoré. -- Comportement de rotation : OmniRoute achemine automatiquement vers le prochain compte Codex éligible. -- Comportement de réinitialisation : lorsque le délai "resetAt" du fournisseur est écoulé, le compte redevient automatiquement éligible. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Scénarios : +Scenarios: -- `5h ON` + `Weekly ON` : le compte est ignoré lorsque l'une ou l'autre des fenêtres atteint le seuil. -- `5h OFF` + `Weekly ON` : seule une utilisation hebdomadaire peut bloquer le compte. -- `5h ON` + `Weekly OFF` : seule une utilisation de 5 heures peut bloquer le compte. -- `resetAt` réussi : le compte entre à nouveau automatiquement dans la rotation (pas de réactivation manuelle).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1444,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Meilleur rapport qualité-prix :**Énorme niveau gratuit ! Utilisez-le avant les niveaux payants.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1459,74 +1662,91 @@ Models:
- +
+🔑 API Key Providers -🔑 Fournisseurs de clés API### NVIDIA NIM (FREE developer access — 70+ models) +### NVIDIA NIM (FREE developer access — 70+ models) -1. Inscrivez-vous : [build.nvidia.com](https://build.nvidia.com) -2. Obtenez une clé API gratuite (1 000 crédits d'inférence inclus) -3. Tableau de bord → Ajouter un fournisseur → NVIDIA NIM : - - Clé API : `nvapi-votre-clé` +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Modèles :**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` et plus de 50 autres +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -**Conseil de pro :**API compatible OpenAI — fonctionne de manière transparente avec la traduction de format d'OmniRoute !### DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -1. Inscrivez-vous : [platform.deepseek.com](https://platform.deepseek.com) -2. Obtenez la clé API -3. Tableau de bord → Ajouter un fournisseur → DeepSeek +### DeepSeek -**Modèles :**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -1. Inscrivez-vous : [console.groq.com](https://console.groq.com) -2. Obtenez la clé API (niveau gratuit inclus) -3. Tableau de bord → Ajouter un fournisseur → Groq +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Modèles :**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +### Groq (Free Tier Available!) -**Conseil de pro :**Inférence ultra-rapide : idéale pour le codage en temps réel !### OpenRouter (100+ Models) +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -1. Inscrivez-vous : [openrouter.ai](https://openrouter.ai) -2. Obtenez la clé API -3. Tableau de bord → Ajouter un fournisseur → OpenRouter +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Modèles :**Accédez à plus de 100 modèles de tous les principaux fournisseurs via une seule clé API. +**Pro Tip:** Ultra-fast inference — best for real-time coding! -**Comportement du tableau de bord :**Les modèles OpenRouter sont gérés à partir des**Modèles disponibles**. L'ajout manuel, l'importation et la synchronisation automatique mettent tous à jour la même liste.
+### OpenRouter (100+ Models) - +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -💰 Fournisseurs bon marché (sauvegarde)### GLM-4.7 (Daily reset, $0.6/1M) +**Models:** Access 100+ models from all major providers through a single API key. -1. Inscrivez-vous : [Zhipu AI](https://open.bigmodel.cn/) -2. Obtenez la clé API du plan de codage -3. Tableau de bord → Ajouter une clé API : - - Fournisseur : `glm` - - Clé API : `votre-clé` +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -**Utilisez :**`glm/glm-4.7` + -**Conseil de pro :**Le plan de codage offre un quota de 3 × à un coût de 1/7 ! Réinitialisation quotidienne à 10h00.### MiniMax M2.1 (5h reset, $0.20/1M) +
+💰 Cheap Providers (Backup) -1. Inscrivez-vous : [MiniMax](https://www.minimax.io/) -2. Obtenez la clé API -3. Tableau de bord → Ajouter une clé API +### GLM-4.7 (Daily reset, $0.6/1M) -**Utilisez :**`minimax/MiniMax-M2.1` +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**Conseil de pro :**Option la moins chère pour un contexte long (1 million de jetons) !### Kimi K2 ($9/month flat) +**Use:** `glm/glm-4.7` -1. Abonnez-vous : [Moonshot AI](https://platform.moonshot.ai/) -2. Obtenez la clé API -3. Tableau de bord → Ajouter une clé API +**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. -**Utilisez :**`kimi/kimi-latest` +### MiniMax M2.1 (5h reset, $0.20/1M) -**Conseil de pro :**Fixe à 9 $/mois pour 10 millions de jetons = 0,90 $/1 M de coût effectif !
+1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key - +**Use:** `minimax/MiniMax-M2.1` -🆓 Fournisseurs GRATUITS (sauvegarde d'urgence)### Qoder (5 FREE models via OAuth) +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1567,9 +1787,10 @@ Models:
- +
+🎨 Create Combos -🎨 Créer des combos### Example 1: Maximize Subscription → Cheap Backup +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1597,9 +1818,10 @@ Cost: $0 forever!
- +
+🔧 CLI Integration -🔧 Intégration CLI### Cursor IDE +### Cursor IDE ``` Settings → Models → Advanced: @@ -1610,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Utilisez la page**CLI Tools**dans le tableau de bord pour une configuration en un clic, ou modifiez manuellement `~/.claude/settings.json`.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1621,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Option 1 — Tableau de bord (recommandé) :**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Option 2 — Manuel :**Modifiez `~/.openclaw/openclaw.json` :```json +```json { "models": { "providers": { @@ -1638,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Remarque :**OpenClaw ne fonctionne qu'avec OmniRoute local. Utilisez « 127.0.0.1 » au lieu de « localhost » pour éviter les problèmes de résolution IPv6.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1652,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Étape 1 :**Ajoutez OmniRoute en tant que fournisseur personnalisé :```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Étape 2 :**Créez/modifiez « opencode.json » à la racine de votre projet :```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1678,117 +1909,130 @@ opencode } } } -```` +``` -**Étape 3 :**Sélectionnez le modèle dans OpenCode :```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Conseil :**Ajoutez n'importe quel modèle disponible dans votre point de terminaison OmniRoute `/v1/models` à la section `models`. Utilisez le format « fournisseur/modèle-id » de votre tableau de bord OmniRoute.
+ --- ## Dépannage - -Cliquez pour développer le guide de dépannage +
+Click to expand troubleshooting guide -**"Le modèle linguistique n'a pas fourni de messages"** +**"Language model did not provide messages"** -- Quota du fournisseur épuisé → Vérifier le suivi des quotas du tableau de bord -- Solution : utilisez la solution de secours combinée ou passez à un niveau moins cher +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier **Rate limiting** -- Quota d'abonnement épuisé → Repli vers GLM/MiniMax -- Ajouter un combo : `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` **OAuth token expired** -- Actualisé automatiquement par OmniRoute -- Si les problèmes persistent : Tableau de bord → Fournisseur → Reconnecter +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect **High costs** -- Vérifiez les statistiques d'utilisation dans le tableau de bord → Coûts -- Passer du modèle principal à GLM/MiniMax -- Utilisez le niveau gratuit (Gemini CLI, Qoder) pour les tâches non critiques +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Les ports du tableau de bord/API sont incorrects** +**Dashboard/API ports are wrong** -- `PORT` est le port de base canonique (et le port API par défaut) -- `API_PORT` remplace uniquement l'écouteur d'API compatible OpenAI -- `DASHBOARD_PORT` remplace uniquement l'écouteur du tableau de bord/Next.js -- Définissez `NEXT_PUBLIC_BASE_URL` sur votre tableau de bord/URL publique (pour les rappels OAuth) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) **Cloud sync errors** -- Vérifiez que `BASE_URL` pointe vers votre instance en cours d'exécution -- Vérifiez que « CLOUD_URL » pointe vers votre point de terminaison cloud attendu -- Gardez les valeurs `NEXT_PUBLIC_*` alignées avec les valeurs côté serveur +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**La première connexion ne fonctionne pas** +**First login not working** -- Vérifiez `INITIAL_PASSWORD` dans `.env` -- S'il n'est pas défini, le mot de passe de secours est « 123456 » +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` **No request logs** -- Les artefacts de requête sont écrits dans `DATA_DIR/call_logs/` sous la forme d'un fichier JSON par requête -- Activez la capture du pipeline depuis le tableau de bord → Journaux → Demander des journaux si vous avez besoin de charges utiles détaillées par étape -- Définissez `APP_LOG_TO_FILE=true` si vous souhaitez également les journaux de la console d'application dans `logs/application/app.log` -- Ajustez `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` et `CALL_LOG_MAX_ENTRIES` selon vos besoins +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Le test de connexion indique « Invalide » pour les fournisseurs compatibles OpenAI** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- De nombreux fournisseurs n'exposent pas de point de terminaison `/models` -- OmniRoute v1.0.6+ inclut une validation de secours via la complétion du chat -- Assurez-vous que l'URL de base inclut le suffixe `/v1`### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ Important pour les utilisateurs exécutant OmniRoute sur un VPS, Docker ou tout autre serveur distant**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -Les fournisseurs**Antigravity**et**Gemini CLI**utilisent**Google OAuth 2.0**. Google exige que le « redirect_uri » dans le flux OAuth corresponde exactement à l'un des URI préenregistrés dans la Google Cloud Console de l'application. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -Les informations d'identification OAuth regroupées dans OmniRoute sont enregistrées**pour `localhost` uniquement**. Lorsque vous accédez à OmniRoute sur un serveur distant (par exemple `https://omniroute.myserver.com`), Google rejette l'authentification avec :``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Vous devez créer un**ID client OAuth 2.0**dans Google Cloud Console avec l'URI de votre serveur.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Ouvrez Google Cloud Console** +#### Step-by-step -Accédez à : [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** + +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) **2. Create a new OAuth 2.0 Client ID** -- Cliquez sur**"+ Créer des informations d'identification"**→**"ID client OAuth"** -- Type d'application :**"Application Web"** -- Nom : tout ce que vous voulez (par exemple "OmniRoute Remote") +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -**3. Ajouter des URI de redirection autorisés** +**3. Add Authorized Redirect URIs** -Dans le champ**"URI de redirection autorisés"**, ajoutez :``` +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Remplacez « votre-serveur.com » par le domaine ou l'IP de votre serveur (incluez le port si nécessaire, par exemple « http://45.33.32.156:20128/callback »). +**4. Save and copy the credentials** -**4. Enregistrez et copiez les informations d'identification** +After creating, Google will show the **Client ID** and **Client Secret**. -Après la création, Google affichera l'**ID client**et le**Secret client**. +**5. Set environment variables** -**5. Définir les variables d'environnement** +In your `.env` (or Docker environment variables): -Dans votre `.env` (ou variables d'environnement Docker) :```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1797,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Restart OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute - -```` +``` **7. Try connecting again** -Tableau de bord → Fournisseurs → Antigravity (ou Gemini CLI) → OAuth +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Google va désormais rediriger correctement vers « https://your-server.com/callback ».--- +Google will now redirect correctly to `https://your-server.com/callback`. + +--- #### Temporary workaround (without custom credentials) -Si vous ne souhaitez pas configurer vos propres identifiants pour le moment, vous pouvez toujours utiliser le**flux d'URL manuel** : +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute ouvre l'URL d'autorisation Google -2. Après autorisation, Google tente de rediriger vers « localhost » (ce qui échoue sur le serveur distant) -3.**Copiez l'URL complète**depuis la barre d'adresse de votre navigateur (même si la page ne se charge pas) -4. Collez cette URL dans le champ affiché dans le modal de connexion OmniRoute. -5. Click**"Connect"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Cela fonctionne car le code d'autorisation dans l'URL est valide, que la page de redirection soit chargée ou non.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. - -🇧🇷 Version en portugais#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Les fournisseurs**Antigravity**et**Gemini CLI**utilisent**Google OAuth 2.0**pour l'authentification. Google exige que `redirect_uri` soit utilisé pour que le flux OAuth soit**exactement**un URI pré-cadastré dans l'application Google Cloud Console. +
+🇧🇷 Versão em Português -Comme les informations d'identification OAuth sont intégrées à OmniRoute, elles sont**spécialisées pour `localhost`**. Lorsque vous accédez à OmniRoute sur un serveur distant (ex : `https://omniroute.meuservidor.com`), Google refuse l'authentification avec :``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Vous devez précisément créer un**ID client OAuth 2.0**dans Google Cloud Console avec l'URI de votre serveur.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Accéder à Google Cloud Console** +#### Passo a passo -Abra : [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Acesse o Google Cloud Console** -**2. Appelez un nouvel ID client OAuth 2.0** +Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Cliquez dessus**"+ Créer des informations d'identification"**→**"ID client OAuth"** -- Type d'application :**"Application Web"** -- Nom : escolha qualquer nome (ex : `OmniRoute Remote`) +**2. Crie um novo OAuth 2.0 Client ID** -**3. Adicione comme URI de redirection autorisés** +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -Pas de champ**"URI de redirection autorisés"**, ajouter :``` +**3. Adicione as Authorized Redirect URIs** + +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Remplacez `seu-servidor.com` par votre domaine ou l'adresse IP de votre serveur (y compris le port si nécessaire, par exemple : `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. Salve et copie comme credenciais** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Après avoir crié, Google affichera le**Client ID**et le**Client Secret**. +**5. Configure as variáveis de ambiente** -**5. Configurer comme variables d'ambiance** +No seu `.env` (ou nas variáveis de ambiente do Docker): -Pas votre `.env` (ou les variables ambiantes de Docker) :```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1876,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reinicie o OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute +``` -```` +**7. Tente conectar novamente** -**7. Tente de connexion nouvelle** +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Tableau de bord → Fournisseurs → Antigravité (ou Gemini CLI) → OAuth +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. -Agora ou Google redirigera directement vers `https://seu-servidor.com/callback` et la fonction d'authentification.--- +--- #### Workaround temporário (sem configurar credenciais próprias) -Si vous ne souhaitez pas créer des informations d'identification appropriées il y a peu, vous pouvez également utiliser le flux**manuel d'URL** : +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. OmniRoute ouvre une URL d'autorisation de Google -2. Après avoir autorisé, Google tente de rediriger vers `localhost` (qui n'est pas un serveur distant) -3.**Copiez une URL complète**à partir de la barre d'adresse de votre navigateur (même si la page n'est pas fermée) -4. Cole est une URL dans le champ qui apparaît dans le modal de connexion à OmniRoute -5. Cliquez dessus**"Connecter"** +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) +4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute +5. Clique em **"Connect"** -> Cette solution de contournement fonctionne parce que le code d'autorisation de l'URL est valide indépendamment de la redirection lorsqu'il est chargé ou non.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1914,64 +2171,73 @@ Si vous ne souhaitez pas créer des informations d'identification appropriées i ## 🛠️ Tech Stack - -Cliquez pour développer les détails de la pile technologique +
+Click to expand tech stack details --**Exécution** : Node.js 18-22 LTS (⚠️ Node.js 24+ n'est**pas pris en charge**— les binaires natifs `better-sqlite3` sont incompatibles) --**Langage** : TypeScript 5.9 —**100 % TypeScript**sur `src/` et `open-sse/` (zéro `any` dans les modules principaux depuis la v2.0) --**Framework** : Next.js 16 + React 19 + Tailwind CSS 4 --**Base de données** : LowDB (JSON) + SQLite (état du domaine + journaux proxy + audit MCP + décisions de routage) --**Schémas** : Zod (validation des E/S de l'outil MCP, contrats API) --**Protocoles** : MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Streaming** : événements envoyés par le serveur (SSE) --**Auth** : OAuth 2.0 (PKCE) + JWT + Clés API + Autorisation étendue MCP --**Tests** : lanceur de tests Node.js + Vitest (900+ tests incluant unitaire, intégration, E2E) --**CI/CD** : actions GitHub (publication automatique npm + Docker Hub à la sortie) --**Site Internet**: [omniroute.online](https://omniroute.online) --**Package** : [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker** : [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Résilience** : disjoncteur, interruption exponentielle, troupeau anti-tonnerre, usurpation d'identité TLS, auto-réparation automatique
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Documentation -| Documenter | Descriptif | +| Document | Description | | ---------------------------------------------- | --------------------------------------------------- | -| [Guide de l'utilisateur](docs/USER_GUIDE.md) | Fournisseurs, combos, intégration CLI, déploiement | -| [Référence API](docs/API_REFERENCE.md) | Tous les points de terminaison avec des exemples | -| [Serveur MCP](open-sse/mcp-server/README.md) | 16 outils MCP, configurations IDE, clients Python/TS/Go | -| [Serveur A2A](src/lib/a2a/README.md) | Protocole JSON-RPC 2.0, compétences, streaming, gestion des tâches | -| [Moteur Auto-Combo](docs/auto-combo.md) | Score à 6 facteurs, packs de modes, auto-guérison | -| [Dépannage](docs/TROUBLESHOOTING.md) | Problèmes courants et solutions | -| [Architecture](docs/ARCHITECTURE.md) | Architecture du système et composants internes | -| [Contribuer](CONTRIBUTING.md) | Configuration et directives de développement | -| [Spécifications OpenAPI](docs/openapi.yaml) | Spécification OpenAPI 3.0 | -| [Politique de sécurité](SECURITY.md) | Rapports de vulnérabilité et pratiques de sécurité | -| [Déploiement de VM](docs/VM_DEPLOYMENT_GUIDE.md) | Guide complet : configuration VM + nginx + Cloudflare | -| [Galerie de fonctionnalités](docs/FEATURES.md) | Visite visuelle du tableau de bord avec captures d'écran | -| [Liste de contrôle de publication](docs/RELEASE_CHECKLIST.md) | Étapes de validation avant la publication |--- +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute propose**plus de 210 fonctionnalités prévues**au cours de plusieurs phases de développement. Here are the key areas: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Catégorie | Planned Features | Faits saillants | -| ----------------------------- | ---------------- | ---------------------------------------------------------------------------- | -| 🧠**Routage & Intelligence**| 25+ | Routage avec la latence la plus faible, routage basé sur des balises, contrôle en amont des quotas, sélection de comptes P2C | -| 🔒**Sécurité et conformité**| 20+ | Renforcement SSRF, masquage des informations d'identification, limite de débit par point de terminaison, portée des clés de gestion | -| 📊**Observabilité**| 15+ | Intégration OpenTelemetry, surveillance des quotas en temps réel, suivi des coûts par modèle | -| 🔄**Intégrations de fournisseurs**| 20+ | Registre de modèles dynamique, temps de recharge des fournisseurs, Codex multi-comptes, analyse des quotas Copilot | -| ⚡**Performances**| 15+ | Double couche de cache, cache d'invite, cache de réponse, streaming keepalive, API par lots | -| 🌐**Écosystème**| 10+ | API WebSocket, rechargement à chaud de la configuration, magasin de configuration distribué, mode commercial |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**Intégration OpenCode**— Prise en charge par le fournisseur natif pour l'IDE de codage OpenCode AI -- 🔗**Intégration TRAE**— Prise en charge complète du cadre de développement TRAE AI -- 📦**Batch API**— Traitement par lots asynchrone pour les demandes groupées -- 🎯**Routage basé sur des balises**— Acheminez les requêtes en fonction de balises personnalisées et de métadonnées -- 💰**Stratégie du coût le plus bas**— Sélectionnez automatiquement le fournisseur disponible le moins cher +### 🔜 Coming Soon -> 📝 Spécifications complètes des fonctionnalités disponibles dans [`docs/new-features/`](docs/new-features/) (217 spécifications détaillées)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1980,17 +2246,19 @@ OmniRoute propose**plus de 210 fonctionnalités prévues**au cours de plusieurs ### How to Contribute 1. Fork the repository -2. Créez votre branche de fonctionnalités (`git checkout -b feature/amazing-feature`) -3. Validez vos modifications (`git commit -m 'Ajouter une fonctionnalité étonnante'`) -4. Poussez vers la branche (`git push origin feature/amazing-feature`) +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) 5. Open a Pull Request -Voir [CONTRIBUTING.md](CONTRIBUTING.md) pour des directives détaillées.### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -2002,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Un merci spécial à**[9router](https://github.com/decolua/9router)**de**[decolua](https://github.com/decolua)**— le projet original qui a inspiré ce fork. OmniRoute s'appuie sur cette incroyable base avec des fonctionnalités supplémentaires, des API multimodales et une réécriture complète de TypeScript. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Un merci spécial à**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— l'implémentation Go originale qui a inspiré ce port JavaScript.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Licence -Licence MIT - voir [LICENSE](LICENSE) pour plus de détails.--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/fr/docs/ARCHITECTURE.md b/docs/i18n/fr/docs/ARCHITECTURE.md index 9697b03066..a6d281af28 100644 --- a/docs/i18n/fr/docs/ARCHITECTURE.md +++ b/docs/i18n/fr/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Dernière mise à jour : 2026-03-28_## Executive Summary -OmniRoute est une passerelle de routage d'IA locale et un tableau de bord construit sur Next.js. -Il fournit un seul point de terminaison compatible OpenAI (`/v1/*`) et achemine le trafic vers plusieurs fournisseurs en amont avec traduction, secours, actualisation des jetons et suivi de l'utilisation. -Capacités de base : +_Last updated: 2026-03-28_ -- Surface API compatible OpenAI pour CLI/outils (28 fournisseurs) -- Traduction des requêtes/réponses dans tous les formats de fournisseurs -- Modèle de repli combo (séquence multi-modèles) -- Repli au niveau du compte (multi-comptes par fournisseur) -- Gestion des connexions du fournisseur de clé OAuth + API -- Génération d'embarquement via `/v1/embeddings` (6 fournisseurs, 9 modèles) -- Génération d'images via `/v1/images/generations` (4 fournisseurs, 9 modèles) -- Pensez à l'analyse des balises (`...`) pour les modèles de raisonnement -- Désinfection des réponses pour une compatibilité stricte avec le SDK OpenAI -- Normalisation des rôles (développeur → système, système → utilisateur) pour une compatibilité entre fournisseurs -- Conversion de sortie structurée (json_schema → Gemini ResponseSchema) -- Persistance locale pour les fournisseurs, les clés, les alias, les combos, les paramètres, les prix -- Suivi de l'utilisation/des coûts et journalisation des demandes -- Synchronisation cloud en option pour la synchronisation multi-appareils/états -- Liste d'autorisation/liste de blocage IP pour le contrôle d'accès aux API -- Penser la gestion budgétaire (passthrough/auto/custom/adaptatif) - -Injection rapide du système global -- Suivi de session et prise d'empreintes digitales -- Limitation de débit améliorée par compte avec des profils spécifiques au fournisseur -- Modèle de disjoncteur pour la résilience du fournisseur -- Protection de troupeau anti-tonnerre avec verrouillage mutex -- Cache de déduplication de requêtes basé sur les signatures -- Couche domaine : disponibilité du modèle, règles de coûts, politique de repli, politique de verrouillage -- Persistance de l'état du domaine (cache en écriture SQLite pour les solutions de repli, les budgets, les verrouillages, les disjoncteurs) -- Moteur de politique pour l'évaluation centralisée des demandes (verrouillage → budget → repli) -- Demande de télémétrie avec agrégation de latence p50/p95/p99 -- ID de corrélation (X-Request-Id) pour le traçage de bout en bout -- Journalisation d'audit de conformité avec désinscription par clé API -- Cadre d'évaluation pour l'assurance qualité LLM -- Tableau de bord de l'interface utilisateur de résilience avec l'état du disjoncteur en temps réel -- Fournisseurs OAuth modulaires (12 modules individuels sous `src/lib/oauth/providers/`) +## Executive Summary -Modèle d'exécution principal : +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- Les routes de l'application Next.js sous `src/app/api/*` implémentent à la fois les API de tableau de bord et les API de compatibilité -- Un noyau SSE/routage partagé dans `src/sse/*` + `open-sse/*` gère l'exécution, la traduction, le streaming, le repli et l'utilisation du fournisseur## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Runtime de la passerelle locale -- API de gestion des tableaux de bord -- Authentification du fournisseur et actualisation du jeton -- Demander une traduction et un streaming SSE -- État local + persistance d'utilisation -- Orchestration de synchronisation cloud en option### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Implémentation du service Cloud derrière `NEXT_PUBLIC_CLOUD_URL` -- SLA/plan de contrôle du fournisseur en dehors du processus local -- Les binaires CLI externes eux-mêmes (Claude CLI, Codex CLI, etc.)## Dashboard Surface (Current) +### Out of Scope -Pages principales sous `src/app/(dashboard)/dashboard/` : +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — démarrage rapide + aperçu du fournisseur -- `/dashboard/endpoint` — proxy de point de terminaison + MCP + A2A + onglets de point de terminaison API -- `/dashboard/providers` — connexions et informations d'identification du fournisseur -- `/dashboard/combos` — stratégies de combo, modèles, règles de routage de modèles -- `/dashboard/costs` — agrégation des coûts et visibilité sur les prix -- `/dashboard/analytics` — analyses et évaluations d'utilisation -- `/dashboard/limits` — contrôles de quotas/taux -- `/dashboard/cli-tools` — Intégration CLI, détection d'exécution, génération de configuration -- `/dashboard/agents` — agents ACP détectés + enregistrement d'agent personnalisé -- `/dashboard/media` — terrain de jeu image/vidéo/musique -- `/dashboard/search-tools` — tests et historique du moteur de recherche -- `/dashboard/health` — temps de disponibilité, disjoncteurs, limites de débit -- `/dashboard/logs` — journaux de requête/proxy/audit/console -- `/dashboard/settings` — onglets des paramètres système (général, routage, valeurs par défaut des combos, etc.) -- `/dashboard/api-manager` — Cycle de vie des clés API et autorisations du modèle## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Principaux répertoires : +Main directories: -- `src/app/api/v1/*` et `src/app/api/v1beta/*` pour les API de compatibilité -- `src/app/api/*` pour les API de gestion/configuration -- Les réécritures suivantes dans `next.config.mjs` mappent `/v1/*` en `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Itinéraires de compatibilité importants : +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` — inclut des modèles personnalisés avec `custom: true` -- `src/app/api/v1/embeddings/route.ts` — génération d'intégration (6 fournisseurs) -- `src/app/api/v1/images/generations/route.ts` — génération d'images (4+ fournisseurs dont Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — chat dédié par fournisseur -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — intégrations dédiées par fournisseur -- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — images dédiées par fournisseur +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` -- `src/app/api/v1beta/models/[...chemin]/route.ts` +- `src/app/api/v1beta/models/[...path]/route.ts` -Domaines de gestion : +Management domains: -- Authentification/paramètres : `src/app/api/auth/*`, `src/app/api/settings/*` -- Fournisseurs/connexions : `src/app/api/providers*` -- Nœuds fournisseurs : `src/app/api/provider-nodes*` -- Modèles personnalisés : `src/app/api/provider-models` (GET/POST/DELETE) -- Catalogue de modèles : `src/app/api/models/route.ts` (GET) -- Configuration du proxy : `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - -OAuth : `src/app/api/oauth/*` -- Clés/alias/combos/pricing : `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- Utilisation : `src/app/api/usage/*` -- Synchronisation/cloud : `src/app/api/sync/*`, `src/app/api/cloud/*` -- Aides aux outils CLI : `src/app/api/cli-tools/*` -- Filtre IP : `src/app/api/settings/ip-filter` (GET/PUT) -- Budget de réflexion : `src/app/api/settings/thinking-budget` (GET/PUT) -- Invite système : `src/app/api/settings/system-prompt` (GET/PUT) -- Sessions : `src/app/api/sessions` (GET) -- Limites de débit : `src/app/api/rate-limits` (GET) -- Résilience : `src/app/api/resilience` (GET/PATCH) — profils de fournisseur, disjoncteur, état limite de débit -- Réinitialisation de la résilience : `src/app/api/resilience/reset` (POST) — réinitialisation des disjoncteurs + temps de recharge -- Statistiques du cache : `src/app/api/cache/stats` (GET/DELETE) -- Disponibilité du modèle : `src/app/api/models/availability` (GET/POST) -- Télémétrie : `src/app/api/telemetry/summary` (GET) -- Budget : `src/app/api/usage/budget` (GET/POST) -- Chaînes de secours : `src/app/api/fallback/chains` (GET/POST/DELETE) -- Audit de conformité : `src/app/api/compliance/audit-log` (GET) -- Évaluations : `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- Politiques : `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- OAuth: `src/app/api/oauth/*` +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -Principaux modules de flux : +## 2) SSE + Translation Core -- Entrée : `src/sse/handlers/chat.ts` -- Orchestration de base : `open-sse/handlers/chatCore.ts` -- Adaptateurs d'exécution du fournisseur : `open-sse/executors/*` -- Détection de format/configuration du fournisseur : `open-sse/services/provider.ts` -- Analyse/résolution du modèle : `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Logique de repli du compte : `open-sse/services/accountFallback.ts` -- Registre de traduction : `open-sse/translator/index.ts` -- Transformations de flux : `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- Extraction/normalisation d'utilisation : `open-sse/utils/usageTracking.ts` -- Analyseur de balises Think : `open-sse/utils/thinkTagParser.ts` -- Gestionnaire d'intégration : `open-sse/handlers/embeddings.ts` -- Registre du fournisseur d'intégration : `open-sse/config/embeddingRegistry.ts` -- Gestionnaire de génération d'images : `open-sse/handlers/imageGeneration.ts` -- Registre du fournisseur d'images : `open-sse/config/imageRegistry.ts` -- Désinfection des réponses : `open-sse/handlers/responseSanitizer.ts` -- Normalisation des rôles : `open-sse/services/roleNormalizer.ts` +Main flow modules: -Services (logique métier) : +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- Sélection/scoration des comptes : `open-sse/services/accountSelector.ts` -- Gestion du cycle de vie du contexte : `open-sse/services/contextManager.ts` -- Application du filtre IP : `open-sse/services/ipFilter.ts` -- Suivi de session : `open-sse/services/sessionManager.ts` -- Demande de déduplication : `open-sse/services/signatureCache.ts` -- Injection d'invite système : `open-sse/services/systemPrompt.ts` -- Penser la gestion budgétaire : `open-sse/services/thinkingBudget.ts` -- Routage du modèle générique : `open-sse/services/wildcardRouter.ts` -- Gestion des limites de débit : `open-sse/services/rateLimitManager.ts` -- Disjoncteur : `open-sse/services/circuitBreaker.ts` +Services (business logic): -Modules de couche de domaine : +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- Disponibilité du modèle : `src/lib/domain/modelAvailability.ts` -- Règles de coûts/budgets : `src/lib/domain/costRules.ts` -- Politique de repli : `src/lib/domain/fallbackPolicy.ts` -- Résolveur combo : `src/lib/domain/comboResolver.ts` -- Politique de verrouillage : `src/lib/domain/lockoutPolicy.ts` -- Moteur de politique : `src/domain/policyEngine.ts` — verrouillage centralisé → budget → évaluation de secours -- Catalogue de codes d'erreur : `src/lib/domain/errorCodes.ts` -- ID de demande : `src/lib/domain/requestId.ts` -- Délai d'expiration de la récupération : `src/lib/domain/fetchTimeout.ts` -- Demande de télémétrie : `src/lib/domain/requestTelemetry.ts` -- Conformité/audit : `src/lib/domain/compliance/index.ts` -- Exécuteur d'évaluation : `src/lib/domain/evalRunner.ts` -- Persistance de l'état du domaine : `src/lib/db/domainState.ts` — SQLite CRUD pour les chaînes de secours, les budgets, l'historique des coûts, l'état de verrouillage, les disjoncteurs +Domain layer modules: -Modules du fournisseur OAuth (12 fichiers individuels sous `src/lib/oauth/providers/`) : +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -- Index du registre : `src/lib/oauth/providers/index.ts` -- Fournisseurs individuels : `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` -- Thin wrapper : `src/lib/oauth/providers.ts` — réexportations à partir de modules individuels## 3) Persistence Layer +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -Base de données d'état primaire (SQLite) : +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -- Infrastructure de base : `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) -- Façade de réexportation : `src/lib/localDb.ts` (fine couche de compatibilité pour les appelants) -- fichier : `${DATA_DIR}/storage.sqlite` (ou `$XDG_CONFIG_HOME/omniroute/storage.sqlite` lorsqu'il est défini, sinon `~/.omniroute/storage.sqlite`) -- entités (tables + espaces de noms KV) : ProviderConnections, ProviderNodes, modelAliases, combos, apiKeys, settings, pricing,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +## 3) Persistence Layer -Persistance d'utilisation : +Primary state DB (SQLite): -- façade : `src/lib/usageDb.ts` (modules décomposés dans `src/lib/usage/*`) -- Tables SQLite dans `storage.sqlite` : `usage_history`, `call_logs`, `proxy_logs` -- des artefacts de fichiers facultatifs restent pour la compatibilité/débogage (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- Les anciens fichiers JSON sont migrés vers SQLite par les migrations de démarrage lorsqu'ils sont présents +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -Base de données d'état du domaine (SQLite) : +Usage persistence: -- `src/lib/db/domainState.ts` — Opérations CRUD pour l'état du domaine -- Tableaux (créés dans `src/lib/db/core.ts`) : `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- Modèle de cache en écriture : les cartes en mémoire font autorité au moment de l'exécution ; les mutations sont écrites de manière synchrone dans SQLite ; l'état est restauré à partir de la base de données lors d'un démarrage à froid## 4) Auth + Security Surfaces +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- Authentification des cookies du tableau de bord : `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- Génération/vérification de clé API : `src/shared/utils/apiKey.ts` -- Les secrets du fournisseur ont persisté dans les entrées `providerConnections` -- Prise en charge du proxy sortant via `open-sse/utils/proxyFetch.ts` (vars env) et `open-sse/utils/networkProxy.ts` (configurable par fournisseur ou global)## 5) Cloud Sync +Domain State DB (SQLite): -- Initialisation du planificateur : `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Tâche périodique : `src/shared/services/cloudSyncScheduler.ts` -- Tâche périodique : `src/shared/services/modelSyncScheduler.ts` -- Route de contrôle : `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Les décisions de secours sont pilotées par « open-sse/services/accountFallback.ts » à l'aide de codes d'état et d'heuristiques de messages d'erreur. Le routage combiné ajoute une protection supplémentaire : les 400 à l'échelle du fournisseur, tels que les échecs de bloc de contenu en amont et de validation de rôle, sont traités comme des échecs locaux du modèle afin que les cibles combinées ultérieures puissent toujours s'exécuter.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -L'actualisation pendant le trafic en direct est exécutée dans `open-sse/handlers/chatCore.ts` via l'exécuteur `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -La synchronisation périodique est déclenchée par « CloudSyncScheduler » lorsque le cloud est activé.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -Fichiers de stockage physique : +Physical storage files: -- Base de données d'exécution principale : `${DATA_DIR}/storage.sqlite` -- lignes de journal de requête : `${DATA_DIR}/log.txt` (artefact compat/debug) -- archives de charge utile d'appel structurées : `${DATA_DIR}/call_logs/` -- sessions facultatives de débogage de traduction/demande : `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*` : API de compatibilité -- `src/app/api/v1/providers/[provider]/*` : routes dédiées par fournisseur (chat, intégrations, images) -- `src/app/api/providers*` : fournisseur CRUD, validation, tests -- `src/app/api/provider-nodes*` : gestion des nœuds compatibles personnalisés -- `src/app/api/provider-models` : gestion de modèles personnalisés (CRUD) -- `src/app/api/models/route.ts` : API de catalogue de modèles (alias + modèles personnalisés) -- `src/app/api/oauth/*` : flux OAuth/device-code -- `src/app/api/keys*` : cycle de vie de la clé API locale -- `src/app/api/models/alias` : gestion des alias -- `src/app/api/combos*` : gestion des combos de repli -- `src/app/api/pricing` : remplacements de prix pour le calcul des coûts -- `src/app/api/settings/proxy` : configuration du proxy (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test` : test de connectivité proxy sortant (POST) -- `src/app/api/usage/*` : API d'utilisation et de logs -- `src/app/api/sync/*` + `src/app/api/cloud/*` : synchronisation cloud et assistants orientés cloud -- `src/app/api/cli-tools/*` : rédacteurs/vérificateurs de configuration CLI locaux -- `src/app/api/settings/ip-filter` : liste autorisée/liste de blocage IP (GET/PUT) -- `src/app/api/settings/thinking-budget` : configuration du budget du jeton de réflexion (GET/PUT) -- `src/app/api/settings/system-prompt` : invite système globale (GET/PUT) -- `src/app/api/sessions` : liste des sessions actives (GET) -- `src/app/api/rate-limits` : statut de limite de débit par compte (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts` : analyse des requêtes, gestion des combos, boucle de sélection de compte -- `open-sse/handlers/chatCore.ts` : traduction, envoi de l'exécuteur, gestion des nouvelles tentatives/actualisations, configuration du flux -- `open-sse/executors/*` : comportement du réseau et du format spécifique au fournisseur### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts` : registre et orchestration du traducteur -- Demander des traducteurs : `open-sse/translator/request/*` -- Traducteurs de réponses : `open-sse/translator/response/*` -- Constantes de format : `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*` : configuration/état persistant et persistance du domaine sur SQLite -- `src/lib/localDb.ts` : réexportation de compatibilité pour les modules DB -- `src/lib/usageDb.ts` : façade historique d'utilisation/journaux d'appels au-dessus des tables SQLite## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Chaque fournisseur dispose d'un exécuteur spécialisé étendant `BaseExecutor` (dans `open-sse/executors/base.ts`), qui fournit la construction d'URL, la construction d'en-tête, les nouvelles tentatives avec intervalle exponentiel, les hooks d'actualisation des informations d'identification et la méthode d'orchestration `execute()`. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Exécuteur testamentaire | Fournisseur(s) | Manutention spéciale | -| ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------------------------------------------------- | -| `Exécuteur par défaut` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Configuration dynamique d'URL/d'en-tête par fournisseur | -| `AntigravityExecutor` | Google Antigravité | ID de projet/session personnalisés, analyse réessayée après | -| `CodexExecutor` | Codex OpenAI | Injecte des instructions système, force un effort de raisonnement | -| `CurseurExécuteur` | Curseur IDE | Protocole ConnectRPC, encodage Protobuf, signature de demande via somme de contrôle | -| `GithubExecutor` | Copilote GitHub | Actualisation du jeton Copilot, en-têtes imitant VSCode | -| `KiroExécuteur` | AWS CodeWhisperer/Kiro | Format binaire AWS EventStream → conversion SSE | -| `GeminiCLIEExecutor` | CLI Gémeaux | Cycle d'actualisation du jeton Google OAuth | +### Persistence -Tous les autres fournisseurs (y compris les nœuds compatibles personnalisés) utilisent « DefaultExecutor ».## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Fournisseur | Formater | Authentification | Flux | Hors flux | Actualisation des jetons | API d'utilisation | -| --------------------- | ----------------- | ------------------------------- | ---------------- | --------- | ------------------------ | ---------------------------- | ------------------------------ | -| Claude | Claude | Clé API/OAuth | ✅ | ✅ | ✅ | ⚠️ Administrateur uniquement | -| Gémeaux | Gémeaux | Clé API/OAuth | ✅ | ✅ | ✅ | ⚠️Console Cloud | -| CLI Gémeaux | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️Console Cloud | -| Antigravité | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ API de quota complet | -| OpenAI | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ | -| Codex | réponses ouvertes | OAuth | ✅ forcé | ❌ | ✅ | ✅ Limites de taux | -| Copilote GitHub | ouvert | OAuth + Jeton Copilot | ✅ | ✅ | ✅ | ✅ Instantanés de quotas | -| Curseur | curseur | Somme de contrôle personnalisée | ✅ | ✅ | ❌ | ❌ | -| Kiro | Kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | -| Qwen | ouvert | OAuth | ✅ | ✅ | ✅ | ⚠️ Par demande | -| Qoder | ouvert | OAuth (de base) | ✅ | ✅ | ✅ | ⚠️ Par demande | -| OuvrirRouter | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | Claude | Clé API | ✅ | ✅ | ❌ | ❌ | -| Recherche profonde | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ | -| Groq | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ | -| Mistral | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ | -| Perplexité | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ | -| Ensemble IA | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ | -| IA de feux d'artifice | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ | -| Cérébraux | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ | -| Cohérer | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ | -| NIM NVIDIA | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Les formats sources détectés incluent : +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. -- `openaï` +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | + +All other providers (including custom compatible nodes) use the `DefaultExecutor`. + +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: + +- `openai` - `openai-responses` -- 'Claude' -- `Gémeaux` +- `claude` +- `gemini` -Les formats cibles incluent : +Target formats include: -- Discussion/Réponses OpenAI - -Claude -- Enveloppe Gemini/Gemini-CLI/Antigravité - -Kiro -- Curseur +- OpenAI chat/Responses +- Claude +- Gemini/Gemini-CLI/Antigravity envelope +- Kiro +- Cursor -Les traductions utilisent**OpenAI comme format hub**— toutes les conversions passent par OpenAI comme intermédiaire :``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Les traductions sont sélectionnées dynamiquement en fonction de la forme de la charge utile source et du format cible du fournisseur. +Additional processing layers in the translation pipeline: -Couches de traitement supplémentaires dans le pipeline de traduction : +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Désinfection des réponses**— Supprime les champs non standard des réponses au format OpenAI (à la fois en streaming et hors streaming) pour garantir une stricte conformité au SDK. --**Normalisation des rôles**— Convertit « développeur » → « système » pour les cibles non OpenAI ; fusionne `system` → `user` pour les modèles qui rejettent le rôle système (GLM, ERNIE) --**Think tag extraction**— Analyse les blocs `...` du contenu dans le champ `reasoning_content` --**Sortie structurée**— Convertit OpenAI `response_format.json_schema` en `responseMimeType` + `responseSchema` de Gemini## Supported API Endpoints +## Supported API Endpoints -| Point de terminaison | Formater | Gestionnaire | +| Endpoint | Format | Handler | | -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | -| `POST /v1/chat/complétions` | Chat OpenAI | `src/sse/handlers/chat.ts` | -| `POST /v1/messages` | Messages de Claude | Même gestionnaire (détecté automatiquement) | -| `POST /v1/réponses` | Réponses OpenAI | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/embeddings` | Intégrations OpenAI | `open-sse/handlers/embeddings.ts` | -| `GET /v1/embeddings` | Liste des modèles | Itinéraire API | -| `POST /v1/images/générations` | Images OpenAI | `open-sse/handlers/imageGeneration.ts` | -| `GET /v1/images/générations` | Liste des modèles | Itinéraire API | -| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dédié par fournisseur avec validation du modèle | -| `POST /v1/providers/{provider}/embeddings` | Intégrations OpenAI | Dédié par fournisseur avec validation du modèle | -| `POST /v1/providers/{provider}/images/générations` | Images OpenAI | Dédié par fournisseur avec validation du modèle | -| `POST /v1/messages/count_tokens` | Compte de jetons Claude | Itinéraire API | -| `GET /v1/models` | Liste des modèles OpenAI | Route API (chat + intégration + image + modèles personnalisés) | -| `GET /api/models/catalogue` | Catalogue | Tous les modèles regroupés par fournisseur + type | -| `POST /v1beta/models/*:streamGenerateContent` | Natif des Gémeaux | Itinéraire API | -| `GET/PUT/DELETE /api/settings/proxy` | Configuration du proxy | Configuration du proxy réseau | -| `POST /api/settings/proxy/test` | Connectivité proxy | Point de terminaison du test d’intégrité/de connectivité du proxy | -| `GET/POST/DELETE /api/provider-models` | Modèles de fournisseurs | Métadonnées du modèle de fournisseur soutenant les modèles disponibles personnalisés et gérés |## Bypass Handler +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -Le gestionnaire de contournement (`open-sse/utils/bypassHandler.ts`) intercepte les requêtes « jetables » connues de Claude CLI — pings d'échauffement, extractions de titres et nombre de jetons — et renvoie une**fausse réponse**sans consommer de jetons du fournisseur en amont. Ceci est déclenché uniquement lorsque `User-Agent` contient `claude-cli`.## Request Logger Pipeline +## Bypass Handler -L'enregistreur de requêtes (`open-sse/utils/requestLogger.ts`) fournit un pipeline de journalisation de débogage en 7 étapes, désactivé par défaut, activé via `ENABLE_REQUEST_LOGS=true` :``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Les fichiers sont écrits dans `/logs//` pour chaque session de requête.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- Temps de recharge du compte du fournisseur en cas d'erreurs transitoires/taux/auth. -- repli du compte avant l'échec de la demande -- repli du modèle combiné lorsque le chemin modèle/fournisseur actuel est épuisé## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- pré-vérification et actualisation avec nouvelle tentative pour les fournisseurs actualisables -- Nouvelle tentative 401/403 après tentative d'actualisation dans le chemin principal## 3) Stream Safety +## 2) Token Expiry -- contrôleur de flux prenant en charge la déconnexion -- flux de traduction avec vidage de fin de flux et gestion `[DONE]` -- repli de l'estimation de l'utilisation lorsque les métadonnées d'utilisation du fournisseur sont manquantes## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- des erreurs de synchronisation apparaissent mais l'exécution locale continue -- le planificateur a une logique capable de réessayer, mais l'exécution périodique appelle actuellement une synchronisation à tentative unique par défaut## 5) Data Integrity +## 3) Stream Safety -- Migrations de schéma SQLite et hooks de mise à niveau automatique au démarrage -- chemin de compatibilité de migration JSON → SQLite hérité## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Sources de visibilité d'exécution : +## 4) Cloud Sync Degradation -- les journaux de la console de `src/sse/utils/logger.ts` -- agrégats d'utilisation par requête dans SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- captures de charge utile détaillées en quatre étapes dans SQLite (`request_detail_logs`) lorsque `settings.detailed_logs_enabled=true` -- journal textuel de l'état de la demande dans `log.txt` (facultatif/compat) -- Journaux facultatifs de requêtes/traductions approfondies sous `logs/` lorsque `ENABLE_REQUEST_LOGS=true` -- points de terminaison d'utilisation du tableau de bord (`/api/usage/*`) pour la consommation de l'interface utilisateur +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -La capture détaillée de la charge utile des requêtes stocke jusqu'à quatre étapes de charge utile JSON par appel routé : +## 5) Data Integrity -- demande brute reçue du client -- requête traduite effectivement envoyée en amont -- réponse du fournisseur reconstruite en JSON ; les réponses diffusées en continu sont compactées dans le résumé final ainsi que les métadonnées du flux -- réponse finale du client renvoyée par OmniRoute ; les réponses diffusées en continu sont stockées dans le même formulaire récapitulatif compact## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- Le secret JWT (`JWT_SECRET`) sécurise la vérification/signature des cookies de session du tableau de bord -- L'amorçage du mot de passe initial (`INITIAL_PASSWORD`) doit être explicitement configuré pour le provisionnement de première exécution -- Le secret de la clé API HMAC (`API_KEY_SECRET`) sécurise le format de clé API locale générée -- Les secrets du fournisseur (clés/jetons API) sont conservés dans la base de données locale et doivent être protégés au niveau du système de fichiers -- Les points de terminaison de synchronisation dans le cloud s'appuient sur l'authentification par clé API + la sémantique de l'identifiant de la machine## Environment and Runtime Matrix +## Observability and Operational Signals -Variables d'environnement activement utilisées par le code : +Runtime visibility sources: -- Application/authentification : `JWT_SECRET`, `INITIAL_PASSWORD` -- Stockage : `DATA_DIR` -- Comportement du nœud compatible : `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Remplacement facultatif de la base de stockage (Linux/macOS lorsque `DATA_DIR` n'est pas défini) : `XDG_CONFIG_HOME` -- Hachage de sécurité : `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Journalisation : `ENABLE_REQUEST_LOGS` -- URL de synchronisation/cloud : `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- Proxy sortant : `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` et variantes minuscules -- Indicateurs de fonctionnalités SOCKS5 : `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- Aides de plate-forme/d'exécution (pas de configuration spécifique à l'application) : `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` et `localDb` partagent la même politique de répertoire de base (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) avec la migration des fichiers hérités. -2. `/api/v1/route.ts` délègue au même constructeur de catalogue unifié utilisé par `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) pour éviter la dérive sémantique. -3. L'enregistreur de requêtes écrit les en-têtes/corps complets lorsqu'il est activé ; traiter le répertoire des journaux comme sensible. -4. Le comportement du cloud dépend de l'exactitude de « NEXT_PUBLIC_BASE_URL » et de l'accessibilité du point de terminaison du cloud. -5. Le répertoire `open-sse/` est publié sous le nom `@omniroute/open-sse`**npm workspace package**. Le code source l'importe via `@omniroute/open-sse/...` (résolu par Next.js `transpilePackages`). Les chemins de fichiers dans ce document utilisent toujours le nom de répertoire « open-sse/ » pour des raisons de cohérence. -6. Les graphiques du tableau de bord utilisent**Recharts**(basé sur SVG) pour des visualisations analytiques accessibles et interactives (graphiques à barres d'utilisation du modèle, tableaux de répartition des fournisseurs avec taux de réussite). -7. Les tests E2E utilisent**Playwright**(`tests/e2e/`), exécutés via `npm run test:e2e`. Les tests unitaires utilisent**l'exécuteur de test Node.js**(`tests/unit/`), exécutés via `npm run test:unit`. Le code source sous `src/` est**TypeScript**(`.ts`/`.tsx`) ; l'espace de travail `open-sse/` reste JavaScript (`.js`). -8. La page Paramètres est organisée en 5 onglets : Sécurité, Routage (6 stratégies globales : remplissage en premier, round-robin, p2c, aléatoire, moins utilisé, coût optimisé), Résilience (limites de débit modifiables, disjoncteur, politiques), IA (budget de réflexion, invite système, cache d'invite), Avancé (proxy).## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- Construire à partir des sources : `npm run build` -- Construire l'image Docker : `docker build -t omniroute .` -- Démarrez le service et vérifiez : +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: - `GET /api/settings` - `GET /api/v1/models` -- L'URL de base cible CLI doit être `http://:20128/v1` lorsque `PORT=20128` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/fr/docs/FEATURES.md b/docs/i18n/fr/docs/FEATURES.md index 114a35308c..c6b62030d1 100644 --- a/docs/i18n/fr/docs/FEATURES.md +++ b/docs/i18n/fr/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Guide visuel de chaque section du tableau de bord OmniRoute.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Gérez les connexions des fournisseurs d'IA : fournisseurs OAuth (Claude Code, Codex, Gemini CLI), fournisseurs de clés API (Groq, DeepSeek, OpenRouter) et fournisseurs gratuits (Qoder, Qwen, Kiro). Les comptes Kiro incluent le suivi du solde créditeur : crédits restants, allocation totale et date de renouvellement visibles dans Tableau de bord → Utilisation.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Créez des combinaisons de routage de modèles avec 6 stratégies : prioritaire, pondérée, à tour de rôle, aléatoire, la moins utilisée et optimisée en termes de coûts. Chaque combo enchaîne plusieurs modèles avec un repli automatique et comprend des modèles rapides et des contrôles de préparation.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Analyses d'utilisation complètes avec consommation de jetons, estimations de coûts, cartes thermiques d'activité, graphiques de distribution hebdomadaire et répartitions par fournisseur.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Surveillance en temps réel : disponibilité, mémoire, version, centiles de latence (p50/p95/p99), statistiques du cache et états des disjoncteurs du fournisseur.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Quatre modes de débogage des traductions d'API :**Playground**(convertisseur de format),**Chat Tester**(requêtes en direct),**Test Bench**(tests par lots) et**Live Monitor**(flux en temps réel).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Testez n’importe quel modèle directement depuis le tableau de bord. Sélectionnez le fournisseur, le modèle et le point de terminaison, rédigez des invites avec Monaco Editor, diffusez les réponses en temps réel, abandonnez en cours de route et affichez les métriques de synchronisation.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Thèmes de couleurs personnalisables pour l'ensemble du tableau de bord. Choisissez parmi 7 couleurs prédéfinies (corail, bleu, rouge, vert, violet, orange, cyan) ou créez un thème personnalisé en choisissant n'importe quelle couleur hexadécimale. Prend en charge les modes clair, sombre et système.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Panneau de paramètres complet avec onglets : +Comprehensive settings panel with tabs: --**Général**— Stockage système, gestion des sauvegardes (base de données d'exportation/importation) -**Apparence**— Sélecteur de thème (sombre/clair/système), préréglages de thèmes de couleurs et couleurs personnalisées, visibilité du journal de santé, contrôles de visibilité des éléments de la barre latérale -**Sécurité**— Protection des points de terminaison de l'API, blocage des fournisseurs personnalisés, filtrage IP, informations de session -**Routage**— Alias de modèle, dégradation des tâches en arrière-plan -**Résilience**— Persistance des limites de débit, réglage du disjoncteur, désactivation automatique des comptes interdits, surveillance de l'expiration des fournisseurs -**Avancé**— Remplacements de configuration, piste d'audit de configuration, mode de dégradation de repli![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Configuration en un clic pour les outils de codage d'IA : Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor et Factory Droid. Comprend l'application/la réinitialisation automatisée de la configuration, les profils de connexion et le mappage de modèle.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Tableau de bord pour découvrir et gérer les agents CLI. Affiche une grille de 14 agents intégrés (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) avec : +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Statut de l'installation**— Installé/Introuvable avec détection de version -**Badges de protocole**— stdio, HTTP, etc. -**Agents personnalisés**— Enregistrez n'importe quel outil CLI via un formulaire (nom, binaire, commande de version, arguments de spawn) -**CLI Fingerprint Matching**— Bascule par fournisseur pour faire correspondre les signatures de requête CLI natives, réduisant ainsi le risque d'interdiction tout en préservant l'adresse IP du proxy.--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Générez des images, des vidéos et de la musique à partir du tableau de bord. Prend en charge OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open et MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Journalisation des demandes en temps réel avec filtrage par fournisseur, modèle, compte et clé API. Affiche les codes d'état, l'utilisation des jetons, la latence et les détails de la réponse.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Votre point de terminaison d'API unifié avec répartition des capacités : achèvements de chat, API de réponses, intégrations, génération d'images, reclassement, transcription audio, synthèse vocale, modérations et clés API enregistrées. Intégration de Cloudflare Quick Tunnel et prise en charge du proxy cloud pour l'accès à distance.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -Créez, définissez et révoquez des clés API. Chaque clé peut être limitée à des modèles/fournisseurs spécifiques avec un accès complet ou des autorisations en lecture seule. Gestion visuelle des clés avec suivi de l'utilisation.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Suivi des actions administratives avec filtrage par type d'action, acteur, cible, adresse IP et horodatage. Historique complet des événements de sécurité.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Application de bureau Native Electron pour Windows, macOS et Linux. Exécutez OmniRoute en tant qu'application autonome avec intégration dans la barre d'état système, prise en charge hors ligne, mise à jour automatique et installation en un clic. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Principales caractéristiques : +Key features: -- Sondage de préparation du serveur (pas d'écran vide au démarrage à froid) -- Barre d'état système avec gestion des ports -- Politique de sécurité du contenu -- Verrouillage à instance unique -- Mise à jour automatique au redémarrage -- Interface utilisateur conditionnelle à la plate-forme (feux de signalisation macOS, barre de titre par défaut Windows/Linux) -- Emballage de build Hardened Electron — les « node_modules » liés symboliquement dans le bundle autonome sont détectés et rejetés avant l'empaquetage, évitant ainsi la dépendance d'exécution sur la machine de build (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Voir [`electron/README.md`](../electron/README.md) pour une documentation complète. +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/fr/docs/TROUBLESHOOTING.md b/docs/i18n/fr/docs/TROUBLESHOOTING.md index ef38899445..219c35c693 100644 --- a/docs/i18n/fr/docs/TROUBLESHOOTING.md +++ b/docs/i18n/fr/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -Problèmes courants et solutions pour OmniRoute.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Problème | Solutions | -| ---------------------------------------------- | -------------------------------------------------------------------------------------- | --- | -| La première connexion ne fonctionne pas | Définissez `INITIAL_PASSWORD` dans `.env` (pas de valeur par défaut codée en dur) | -| Le tableau de bord s'ouvre sur le mauvais port | Définissez `PORT=20128` et `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Aucun journal de requête sous `logs/` | Définir `ENABLE_REQUEST_LOGS=true` | -| EACCES : autorisation refusée | Définissez `DATA_DIR=/path/to/writable/dir` pour remplacer `~/.omniroute` | -| La stratégie de routage ne sauvegarde pas | Mise à jour vers v1.4.11+ (correctif du schéma Zod pour la persistance des paramètres) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Cause :**Quota de fournisseur épuisé. +**Cause:** Provider quota exhausted. -**Correction :** +**Fix:** -1. Vérifiez le suivi des quotas du tableau de bord -2. Utilisez un combo avec des niveaux de secours -3. Passez au niveau moins cher/gratuit### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Cause :**Quota d'abonnement épuisé. +### Rate Limiting -**Correction :** +**Cause:** Subscription quota exhausted. -- Ajouter une solution de secours : `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Utilisez GLM/MiniMax comme sauvegarde bon marché### OAuth Token Expired +**Fix:** -OmniRoute actualise automatiquement les jetons. Si les problèmes persistent : +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Tableau de bord → Fournisseur → Reconnecter -2. Supprimez et rajoutez la connexion du fournisseur--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Vérifiez que `BASE_URL` pointe vers votre instance en cours d'exécution (par exemple, `http://localhost:20128`) -2. Vérifiez que « CLOUD_URL » pointe vers votre point de terminaison cloud (par exemple, « https://omniroute.dev ») -3. Gardez les valeurs `NEXT_PUBLIC_*` alignées avec les valeurs côté serveur### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Symptôme :**`Jeton inattendu 'd'...` sur le point de terminaison cloud pour les appels sans streaming. +### Cloud `stream=false` Returns 500 -**Cause :**Upstream renvoie la charge utile SSE alors que le client attend du JSON. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Solution de contournement :**Utilisez « stream=true » pour les appels directs vers le cloud. Le runtime local inclut le repli SSE → JSON.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Créez une nouvelle clé à partir du tableau de bord local (`/api/keys`) -2. Exécutez la synchronisation cloud : Activer le cloud → Synchroniser maintenant -3. Les clés anciennes/non synchronisées peuvent toujours renvoyer « 401 » sur le cloud--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Vérifiez les champs d'exécution : `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. Pour le mode portable : utilisez la cible d'image `runner-cli` (CLI fournies) -3. Pour le mode de montage de l'hôte : définissez `CLI_EXTRA_PATHS` et montez le répertoire bin de l'hôte en lecture seule -4. Si `installed=true` et `runnable=false` : le binaire a été trouvé mais le contrôle de santé a échoué### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Vérifiez les statistiques d'utilisation dans le tableau de bord → Utilisation -2. Basculez le modèle principal vers GLM/MiniMax -3. Utilisez l'offre gratuite (Gemini CLI, Qoder) pour les tâches non critiques -4. Définissez les budgets de coûts par clé API : Tableau de bord → Clés API → Budget--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Définissez `ENABLE_REQUEST_LOGS=true` dans votre fichier `.env`. Les journaux apparaissent dans le répertoire `logs/`.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- État principal : `${DATA_DIR}/storage.sqlite` (fournisseurs, combos, alias, clés, paramètres) -- Utilisation : tables SQLite dans `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + facultatif `${DATA_DIR}/log.txt` et `${DATA_DIR}/call_logs/` -- Journaux de requête : `/logs/...` (quand `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Lorsque le disjoncteur d'un fournisseur est OUVERT, les demandes sont bloquées jusqu'à l'expiration du temps de recharge. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Correction :** +**Fix:** -1. Accédez à**Tableau de bord → Paramètres → Résilience** -2. Vérifiez la carte de disjoncteur du fournisseur concerné -3. Cliquez sur**Réinitialiser tout**pour effacer tous les disjoncteurs ou attendez l'expiration du temps de recharge. -4. Vérifiez que le fournisseur est réellement disponible avant de réinitialiser### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Si un fournisseur entre à plusieurs reprises dans l’état OPEN : +### Provider keeps tripping the circuit breaker -1. Vérifiez**Tableau de bord → Santé → Santé du fournisseur**pour connaître le modèle d'échec. -2. Accédez à**Paramètres → Résilience → Profils de fournisseur**et augmentez le seuil d'échec. -3. Vérifiez si le fournisseur a modifié les limites de l'API ou nécessite une ré-authentification -4. Examinez la télémétrie de latence : une latence élevée peut provoquer des échecs liés au délai d'attente.--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Assurez-vous d'utiliser le préfixe correct : `deepgram/nova-3` ou `assemblyai/best` -- Vérifiez que le fournisseur est connecté dans**Tableau de bord → Fournisseurs**### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Vérifiez les formats audio pris en charge : `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- Vérifiez que la taille du fichier est dans les limites du fournisseur (généralement < 25 Mo) -- Vérifier la validité de la clé API du fournisseur dans la carte du fournisseur--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Utilisez**Tableau de bord → Traducteur**pour déboguer les problèmes de traduction de format : +Use **Dashboard → Translator** to debug format translation issues: -| Mode | Quand utiliser | -| ---------------------- | -------------------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Aire de jeux** | Comparez les formats d'entrée/sortie côte à côte : collez une requête qui a échoué pour voir comment elle se traduit | -| **Testeur de chat** | Envoyez des messages en direct et inspectez la charge utile complète de la demande/réponse, y compris les en-têtes | -| **Banc d'essai** | Exécutez des tests par lots sur les combinaisons de formats pour identifier les traductions défectueuses | -| **Moniteur en direct** | Observez le flux de requêtes en temps réel pour détecter les problèmes de traduction intermittents | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Les balises de réflexion n'apparaissent pas**— Vérifiez si le fournisseur cible prend en charge la réflexion et le paramètre de budget de réflexion -**Abandon des appels d'outils**— Certaines traductions de format peuvent supprimer des champs non pris en charge ; vérifier en mode Playground -**Invite système manquante**— Claude et Gemini gèrent les invites système différemment ; vérifier le résultat de la traduction -**Le SDK renvoie une chaîne brute au lieu d'un objet**— Corrigé dans la version 1.1.0 : le désinfectant de réponse supprime désormais les champs non standard (`x_groq`, `usage_breakdown`, etc.) qui provoquent des échecs de validation OpenAI SDK Pydantic -**GLM/ERNIE rejette le rôle `système`**— Corrigé dans la version 1.1.0 : le normalisateur de rôle fusionne automatiquement les messages système dans les messages utilisateur pour les modèles incompatibles -**Rôle de « développeur » non reconnu**— Corrigé dans la version 1.1.0 : automatiquement converti en « système » pour les fournisseurs non OpenAI -**`json_schema` ne fonctionne pas avec Gemini**— Corrigé dans la v1.1.0 : `response_format` est maintenant converti en `responseMimeType` + `responseSchema` de Gemini--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- La limite de débit automatique s'applique uniquement aux fournisseurs de clés API (pas à OAuth/abonnement) -- Vérifiez que**Paramètres → Résilience → Profils de fournisseur**a activé la limite de débit automatique. -- Vérifiez si le fournisseur renvoie les codes d'état « 429 » ou les en-têtes « Retry-After »### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -Les profils de fournisseur prennent en charge ces paramètres : +### Tuning exponential backoff --**Délai de base**— Temps d'attente initial après le premier échec (par défaut : 1 s) -**Délai maximum**— Limite maximale du temps d'attente (par défaut : 30 s) -**Multiplicateur**— De combien augmenter le délai par échec consécutif (par défaut : 2x)### Anti-thundering herd +Provider profiles support these settings: -Lorsque de nombreuses requêtes simultanées atteignent un fournisseur à débit limité, OmniRoute utilise mutex + limitation de débit automatique pour sérialiser les requêtes et éviter les échecs en cascade. Ceci est automatique pour les fournisseurs de clés API.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Certains utilisateurs d'OmniRoute placent la passerelle devant les RAG ou les piles d'agents. Dans ces configurations, il est courant de voir un schéma étrange : OmniRoute semble sain (fournisseurs activés, profils de routage corrects, aucune alerte de limite de débit) mais la réponse finale est toujours fausse. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -En pratique, ces incidents proviennent généralement du pipeline RAG en aval, et non de la passerelle elle-même. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Si vous souhaitez un vocabulaire partagé pour décrire ces échecs, vous pouvez utiliser le WFGY ProblemMap, une ressource textuelle externe sous licence MIT qui définit seize modèles d'échecs RAG/LLM récurrents. À un niveau élevé, il couvre : +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- dérive de récupération et limites de contexte brisées -- index vides ou obsolètes et magasins de vecteurs -- intégration versus inadéquation sémantique -- problèmes d'assemblage rapide et de fenêtre contextuelle -- effondrement de la logique et réponses trop confiantes -- échecs de la longue chaîne et de la coordination des agents -- mémoire multi-agents et dérive des rôles -- problèmes de déploiement et d'ordre d'amorçage +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -L'idée est simple : +The idea is simple: -1. Lorsque vous enquêtez sur une mauvaise réponse, capturez : - - tâche et demande de l'utilisateur - - combo d'itinéraire ou de fournisseur dans OmniRoute - - tout contexte RAG utilisé en aval (documents récupérés, appels d'outils, etc.) -2. Cartographiez l'incident avec un ou deux numéros WFGY ProblemMap (« No.1 » … « No.16 »). -3. Stockez le numéro dans votre propre tableau de bord, runbook ou suivi des incidents à côté des journaux OmniRoute. -4. Utilisez la page WFGY correspondante pour décider si vous devez modifier votre pile RAG, votre récupérateur ou votre stratégie de routage. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -Texte intégral et recettes concrètes en direct ici (licence MIT, texte uniquement) : +Full text and concrete recipes live here (MIT license, text only): -[WFGY ProblemMap README](https://github.com/onestadao/WFGY/blob/main/ProblemMap/README.md) +[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -Vous pouvez ignorer cette section si vous n'exécutez pas de RAG ou de pipelines d'agent derrière OmniRoute.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**Problèmes GitHub** : [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Architecture** : Voir [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) pour les détails internes -**Référence API** : voir [`docs/API_REFERENCE.md`](API_REFERENCE.md) pour tous les points de terminaison -**Tableau de bord de santé** : consultez**Tableau de bord → Santé**pour connaître l'état du système en temps réel -**Traducteur** : utilisez**Tableau de bord → Traducteur**pour déboguer les problèmes de format +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/fr/llm.txt b/docs/i18n/fr/llm.txt new file mode 100644 index 0000000000..d6223886bf --- /dev/null +++ b/docs/i18n/fr/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Français) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Aperçu + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Sécurité +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/he/README.md b/docs/i18n/he/README.md index e272608e37..3c6c207268 100644 --- a/docs/i18n/he/README.md +++ b/docs/i18n/he/README.md @@ -4,6 +4,7 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. _Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ @@ -247,7 +248,7 @@ Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Eve - **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention - **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI - **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next -- **Custom Combos** — Customizable fallback chains with 9 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random) +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) - **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard @@ -1308,7 +1309,17 @@ Then in `/dashboard/media` → **Transcription** tab: upload any audio or video ## 💡 Key Features -OmniRoute v2.0 is built as an operational platform, not just a relay proxy. +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. + +### 🆕 New — v3.5.5 Highlights (Apr 2026) + +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | ### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) @@ -1360,7 +1371,8 @@ OmniRoute v2.0 is built as an operational platform, not just a relay proxy. | 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | | 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | | 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | -| 🎨 **Custom Combos** | 9 balancing strategies + fallback chain control | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | | 🌐 **Wildcard Router** | `provider/*` dynamic routing | | 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | | 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | @@ -2187,9 +2199,10 @@ Se não quiser criar credenciais próprias agora, ainda é possível usar o flux | ---------------------------------------------- | --------------------------------------------------- | | [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | | [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | -| [MCP Server](open-sse/mcp-server/README.md) | 16 MCP tools, IDE configs, Python/TS/Go clients | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | | [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | | [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | | [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | | [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | | [Contributing](CONTRIBUTING.md) | Development setup and guidelines | diff --git a/docs/i18n/he/docs/ARCHITECTURE.md b/docs/i18n/he/docs/ARCHITECTURE.md index c774f5933d..101c8aca35 100644 --- a/docs/i18n/he/docs/ARCHITECTURE.md +++ b/docs/i18n/he/docs/ARCHITECTURE.md @@ -4,6 +4,8 @@ --- + + _Last updated: 2026-03-28_ ## Executive Summary @@ -36,6 +38,7 @@ Core capabilities: - Anti-thundering herd protection with mutex locking - Signature-based request deduplication cache - Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity - Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) - Policy engine for centralized request evaluation (lockout → budget → fallback) - Request telemetry with p50/p95/p99 latency aggregation @@ -222,6 +225,8 @@ Services (business logic): - Wildcard model routing: `open-sse/services/wildcardRouter.ts` - Rate limit management: `open-sse/services/rateLimitManager.ts` - Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions Domain layer modules: @@ -802,7 +807,10 @@ Environment variables actively used by code: 5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. 6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). 7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). -8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. ## Operational Verification Checklist diff --git a/docs/i18n/he/docs/FEATURES.md b/docs/i18n/he/docs/FEATURES.md index 596e04ad05..2efa645cae 100644 --- a/docs/i18n/he/docs/FEATURES.md +++ b/docs/i18n/he/docs/FEATURES.md @@ -4,6 +4,8 @@ --- + + Visual guide to every section of the OmniRoute dashboard. --- @@ -18,7 +20,7 @@ Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI) ## 🎨 Combos -Create model routing combos with 6 strategies: priority, weighted, round-robin, random, least-used, and cost-optimized. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. ![Combos Dashboard](screenshots/02-combos.png) @@ -68,7 +70,7 @@ Comprehensive settings panel with tabs: - **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls - **Security** — API endpoint protection, custom provider blocking, IP filtering, session info - **Routing** — Model aliases, background task degradation -- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration - **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode ![Settings Dashboard](screenshots/06-settings.png) @@ -94,6 +96,30 @@ Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in a --- +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- + ## 🖼️ Media _(v2.0.3+)_ Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. diff --git a/docs/i18n/he/docs/TROUBLESHOOTING.md b/docs/i18n/he/docs/TROUBLESHOOTING.md index d49dd89c9f..f1f130ce04 100644 --- a/docs/i18n/he/docs/TROUBLESHOOTING.md +++ b/docs/i18n/he/docs/TROUBLESHOOTING.md @@ -4,6 +4,8 @@ --- + + Common problems and solutions for OmniRoute. --- @@ -17,6 +19,60 @@ Common problems and solutions for OmniRoute. | No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | | EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | | Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. --- diff --git a/docs/i18n/he/llm.txt b/docs/i18n/he/llm.txt new file mode 100644 index 0000000000..1eec894ae5 --- /dev/null +++ b/docs/i18n/he/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (עברית) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## סקירה כללית + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### אבטחה +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/hi/README.md b/docs/i18n/hi/README.md index 36549e45b4..bd6ee203d5 100644 --- a/docs/i18n/hi/README.md +++ b/docs/i18n/hi/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_आपका सार्वभौमिक एपीआई प्रॉक्सी - एक समापन बिंदु, 60+ प्रदाता, शून्य डाउनटाइम। अब**एमसीपी सर्वर (25 टूल्स)**,**ए2ए प्रोटोकॉल**,**मेमोरी/स्किल सिस्टम**और**इलेक्ट्रॉन डेस्कटॉप ऐप**के साथ।_ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**चैट समापन • एंबेडिंग • छवि निर्माण • वीडियो • संगीत • ऑडियो • पुनःरैंकिंग •**वेब खोज**• एमसीपी सर्वर • ए2ए प्रोटोकॉल • 100% टाइपस्क्रिप्ट**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _आपका सार्वभौमिक एपीआई प्रॉक् [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 वेबसाइट](https://omniroute.online) • [🚀 त्वरित प्रारंभ](#-त्वरित-प्रारंभ) • [💡 विशेषताएं](#-कुंजी-विशेषताएं) • [📖 दस्तावेज़](#-दस्तावेज़ीकरण) • [💰 मूल्य निर्धारण](#-मूल्य निर्धारण-एक नजर में) • [💬 व्हाट्सएप](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**इसमें उपलब्ध:**🇺🇸 [अंग्रेजी](README.md) | 🇧🇷 [पुर्तगाली (ब्राजील)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [फ़्रांसीसी](docs/i18n/fr/README.md) | 🇮🇹 [इतालवी](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [जर्मन](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [डांस्क](docs/i18n/da/README.md) | 🇫🇮 [सुओमी](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [मग्यार](docs/i18n/hu/README.md) | 🇮🇩 [बहासा इंडोनेशिया](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [बहासा मेलायु](docs/i18n/ms/README.md) | 🇳🇱 [नीदरलैंड्स](docs/i18n/nl/README.md) | 🇳🇴 [नॉर्स्क](docs/i18n/no/README.md) | 🇵🇹 [पुर्तगाली (पुर्तगाल)](docs/i18n/pt/README.md) | 🇷🇴 [रोमानिया](docs/i18n/ro/README.md) | 🇵🇱 [पोल्स्की](docs/i18n/pl/README.md) | 🇸🇰[स्लोवेनसीना](docs/i18n/sk/README.md) | 🇸🇪 [स्वेन्स्का](docs/i18n/sv/README.md) | 🇵🇭 [फ़िलिपिनो](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -54,552 +61,628 @@ _आपका सार्वभौमिक एपीआई प्रॉक् ## 📸 Dashboard Preview
-<सारांश>डैशबोर्ड स्क्रीनशॉट देखने के लिए क्लिक करें +Click to see dashboard screenshots -| पेज | स्क्रीनशॉट | -| ---------------- | ------------------------------------------------------------ | ---------- | -| **प्रदाता** | ![प्रदाता](docs/screenshots/01-providers.png) | -| **कॉम्बोस** | ![कॉम्बोस](दस्तावेज़/स्क्रीनशॉट/02-कॉम्बोस.पीएनजी) | -| **एनालिटिक्स** | ![एनालिटिक्स](docs/screenshots/03-analytics.png) | -| **स्वास्थ्य** | ![स्वास्थ्य](docs/screenshots/04-health.png) | -| **अनुवादक** | ![अनुवादक](docs/screenshots/05-translator.png) | -| **सेटिंग्स** | ![सेटिंग्स](दस्तावेज़/स्क्रीनशॉट/06-सेटिंग्स.पीएनजी) | -| **सीएलआई उपकरण** | ![सीएलआई उपकरण](दस्तावेज़/स्क्रीनशॉट/07-सीएलआई-टूल्स.पीएनजी) | -| **उपयोग लॉग** | ![उपयोग](docs/screenshots/08-usage.png) | -| **अंतबिंदु** | ![समाप्ति बिंदु](दस्तावेज़/स्क्रीनशॉट/09-endpoint.png) |
| +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | + + --- ### 🤖 Free AI Provider for your favorite coding agents -_OmniRoute के माध्यम से किसी भी AI-संचालित IDE या CLI टूल को कनेक्ट करें - असीमित कोडिंग के लिए निःशुल्क API गेटवे।_ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ -<तालिका> - - - -OpenClaw
-ओपनक्लॉ -

-⭐ 205K - - - -NanoBot
-नैनोबॉट -

-⭐ 20.9K - - - -”PicoClaw”
-पिकोक्लॉ -

-⭐ 14.6K - - - -ZeroClaw
-जीरोक्लॉ -

-⭐ 9.9K - - - -IronClaw
-आयरनक्लॉ -

-⭐ 2.1K - - - - - -OpenCode
-ओपनकोड -

-⭐ 106K - - - -कोडेक्स सीएलआई
-कोडेक्स सीएलआई -

-⭐ 60.8K - - - -क्लाउड कोड
-क्लाउड कोड -

-⭐ 67.3K - - - -मिथुन सीएलआई
-मिथुन सीएलआई -

-⭐ 94.7K - - - -किलो कोड
-किलो कोड -

-⭐ 15.5K - - - + + + + + + + + + + + + + + + +
+ + OpenClaw
+ OpenClaw +

+ ⭐ 205K +
+ + NanoBot
+ NanoBot +

+ ⭐ 20.9K +
+ + PicoClaw
+ PicoClaw +

+ ⭐ 14.6K +
+ + ZeroClaw
+ ZeroClaw +

+ ⭐ 9.9K +
+ + IronClaw
+ IronClaw +

+ ⭐ 2.1K +
+ + OpenCode
+ OpenCode +

+ ⭐ 106K +
+ + Codex CLI
+ Codex CLI +

+ ⭐ 60.8K +
+ + Claude Code
+ Claude Code +

+ ⭐ 67.3K +
+ + Gemini CLI
+ Gemini CLI +

+ ⭐ 94.7K +
+ + Kilo Code
+ Kilo Code +

+ ⭐ 15.5K +
-📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**पैसा बर्बाद करना और सीमा पार करना बंद करें:** +**Stop wasting money and hitting limits:** -- सदस्यता कोटा हर महीने अप्रयुक्त रूप से समाप्त हो जाता है -- दर सीमा आपको कोडिंग के बीच में रोक देती है -- महंगे एपीआई ($20-50/माह प्रति प्रदाता) -- प्रदाताओं के बीच मैन्युअल स्विचिंग +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute इसका समाधान करता है:** +**OmniRoute solves this:** -- ✅**सब्सक्रिप्शन अधिकतम करें**- कोटा ट्रैक करें, रीसेट से पहले हर बिट का उपयोग करें -- ✅**ऑटो फ़ॉलबैक**- सदस्यता → एपीआई कुंजी → सस्ता → निःशुल्क, शून्य डाउनटाइम -- ✅**मल्टी-अकाउंट**- प्रति प्रदाता खातों के बीच राउंड-रॉबिन -- ✅**यूनिवर्सल**- क्लाउड कोड, कोडेक्स, जेमिनी सीएलआई, कर्सर, क्लाइन, ओपनक्लॉ, किसी भी सीएलआई टूल के साथ काम करता है--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**हमारे समुदाय में शामिल हों!**[व्हाट्सएप ग्रुप](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) - सहायता प्राप्त करें, टिप्स साझा करें और अपडेट रहें। +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**वेबसाइट**: [omniroute.online](https://omniroute.online) -**गिटहब**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**मुद्दे**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**व्हाट्सएप**: [सामुदायिक समूह](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**योगदान**: [CONTRIBUTING.md](CONTRIBUTING.md) देखें, एक पीआर खोलें, या एक `अच्छा पहला अंक` चुनें -**मूल परियोजना**: [डेकोलुआ द्वारा 9राउटर](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -कोई समस्या खोलते समय, कृपया सिस्टम-जानकारी कमांड चलाएँ और जेनरेट की गई फ़ाइल संलग्न करें:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -यह आपके Node.js संस्करण, ओमनीरूट संस्करण, ओएस विवरण, स्थापित सीएलआई उपकरण (क्यूडर, जेमिनी, क्लाउड, कोडेक्स, एंटीग्रेविटी, ड्रॉइड, आदि), डॉकर/पीएम2 स्थिति और सिस्टम पैकेज के साथ एक `system-info.txt` उत्पन्न करता है - वह सब कुछ जो हमें आपकी समस्या को शीघ्रता से पुन: उत्पन्न करने के लिए चाहिए। फ़ाइल को सीधे अपने GitHub मुद्दे से संलग्न करें।--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**एआई टूल का उपयोग करने वाला प्रत्येक डेवलपर प्रतिदिन इन समस्याओं का सामना करता है।**ओम्नीरूट को उन सभी को हल करने के लिए बनाया गया था - लागत वृद्धि से लेकर क्षेत्रीय ब्लॉक तक, टूटे हुए ओएथ प्रवाह से लेकर प्रोटोकॉल संचालन और एंटरप्राइज़ अवलोकन तक। +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. -<विवरण> -<सारांश>💸 1. "मैं एक महंगी सदस्यता के लिए भुगतान करता हूं लेकिन फिर भी सीमा से बाधित होता हूं" +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -डेवलपर्स क्लाउड प्रो, कोडेक्स प्रो, या गिटहब कोपायलट के लिए $20-200/माह का भुगतान करते हैं। यहां तक ​​कि भुगतान करने पर भी, कोटा की एक सीमा होती है - 5 घंटे का उपयोग, साप्ताहिक सीमा, या प्रति मिनट की दर सीमा। मध्य-कोडिंग सत्र में, प्रदाता प्रत्युत्तर देना बंद कर देता है और डेवलपर प्रवाह और उत्पादकता खो देता है। +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**ओम्नीरूट इसे कैसे हल करता है:** +**How OmniRoute solves it:** --**स्मार्ट 4-टियर फ़ॉलबैक**- यदि सदस्यता कोटा समाप्त हो जाता है, तो स्वचालित रूप से एपीआई कुंजी पर रीडायरेक्ट हो जाता है → सस्ता → शून्य मैन्युअल हस्तक्षेप के साथ मुफ़्त --**प्रदाता ट्रैकिंग सीमाएं**- कैश्ड कोटा स्नैपशॉट सर्वर-साइड शेड्यूल पर रीफ्रेश होता है (डिफ़ॉल्ट `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) यूआई में मैन्युअल रीफ्रेश के साथ उपलब्ध है --**मल्टी-अकाउंट सपोर्ट**- ऑटो राउंड-रॉबिन के साथ प्रति प्रदाता एकाधिक खाते - जब एक खत्म हो जाता है, तो अगले पर स्विच हो जाता है --**कस्टम कॉम्बो**- 9 संतुलन रणनीतियों (प्राथमिकता, भारित, भरण-प्रथम, राउंड-रॉबिन, पी2सी, यादृच्छिक, कम से कम उपयोग, लागत-अनुकूलित, सख्त-यादृच्छिक) के साथ अनुकूलन योग्य फ़ॉलबैक चेन --**कोडेक्स बिजनेस कोटा**- बिजनेस/टीम कार्यक्षेत्र कोटा की निगरानी सीधे डैशबोर्ड में
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard -<विवरण> -<सारांश>🔌 2. "मुझे कई प्रदाताओं का उपयोग करने की आवश्यकता है लेकिन प्रत्येक के पास एक अलग एपीआई है" + -ओपनएआई एक प्रारूप का उपयोग करता है, क्लाउड (एंथ्रोपिक) दूसरे का उपयोग करता है, जेमिनी एक और का उपयोग करता है। यदि कोई डेवलपर विभिन्न प्रदाताओं के मॉडल का परीक्षण करना चाहता है या उनके बीच फ़ॉलबैक करना चाहता है, तो उन्हें एसडीके को फिर से कॉन्फ़िगर करना होगा, एंडपॉइंट बदलना होगा, असंगत प्रारूपों से निपटना होगा। कस्टम प्रदाताओं (फ्रेंडएलआई, एनआईएम) के पास गैर-मानक मॉडल एंडपॉइंट हैं। +
+🔌 2. "I need to use multiple providers but each has a different API" -**ओम्नीरूट इसे कैसे हल करता है:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**एकीकृत समापन बिंदु**- एक एकल `http://localhost:20128/v1` सभी 60+ प्रदाताओं के लिए प्रॉक्सी के रूप में कार्य करता है --**प्रारूप अनुवाद**- स्वचालित और पारदर्शी: ओपनएआई ↔ क्लाउड ↔ जेमिनी ↔ प्रतिक्रिया एपीआई --**प्रतिक्रिया स्वच्छता**- गैर-मानक फ़ील्ड (`x_groq`, `usage_breakdown`, `service_tier`) को हटा दें जो OpenAI SDK v1.83+ को तोड़ता है --**भूमिका सामान्यीकरण**- गैर-ओपनएआई प्रदाताओं के लिए `डेवलपर` → `सिस्टम` को रूपांतरित करता है; GLM/ERNIE के लिए `सिस्टम` → `उपयोगकर्ता` --**थिंक टैग एक्सट्रैक्शन**- डीपसीक आर1 जैसे मॉडलों से `<थिंक>` ब्लॉक को मानकीकृत `रीज़निंग_कंटेंट` में निकाला जाता है --**मिथुन के लिए संरचित आउटपुट**- `json_schema` → `responseMimeType`/`responseSchema` स्वचालित रूपांतरण --**`स्ट्रीम` डिफ़ॉल्ट रूप से `झूठा`**होता है - ओपनएआई स्पेक के साथ संरेखित होता है, पायथन/रस्ट/गो एसडीके में अप्रत्याशित एसएसई से बचता है
+**How OmniRoute solves it:** -<विवरण> -<सारांश>🌐 3. "मेरा AI प्रदाता मेरे क्षेत्र/देश को ब्लॉक कर देता है" +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -OpenAI/Codex जैसे प्रदाता कुछ भौगोलिक क्षेत्रों से पहुंच को रोकते हैं। उपयोगकर्ताओं को OAuth और API कनेक्शन के दौरान `unsupported_country_region_territory` जैसी त्रुटियां मिलती हैं। यह विकासशील देशों के डेवलपर्स के लिए विशेष रूप से निराशाजनक है। + -**ओम्नीरूट इसे कैसे हल करता है:** +
+🌐 3. "My AI provider blocks my region/country" --**3-स्तरीय प्रॉक्सी कॉन्फ़िगरेशन**- 3 स्तरों पर कॉन्फ़िगर करने योग्य प्रॉक्सी: वैश्विक (सभी ट्रैफ़िक), प्रति-प्रदाता (केवल एक प्रदाता), और प्रति-कनेक्शन/कुंजी --**रंग-कोडित प्रॉक्सी बैज**- दृश्य संकेतक: 🟢 वैश्विक प्रॉक्सी, 🟡 प्रदाता प्रॉक्सी, 🔵 कनेक्शन प्रॉक्सी, हमेशा आईपी दिखाता है --**प्रॉक्सी के माध्यम से OAuth टोकन एक्सचेंज**- OAuth प्रवाह भी प्रॉक्सी के माध्यम से जाता है, `unsupported_country_region_territory` को हल करता है --**प्रॉक्सी के माध्यम से कनेक्शन परीक्षण**- कनेक्शन परीक्षण कॉन्फ़िगर प्रॉक्सी का उपयोग करते हैं (अब कोई प्रत्यक्ष बाईपास नहीं) --**SOCKS5 समर्थन**- आउटबाउंड रूटिंग के लिए पूर्ण SOCKS5 प्रॉक्सी समर्थन --**टीएलएस फ़िंगरप्रिंट स्पूफिंग**- बॉट डिटेक्शन को बायपास करने के लिए `wreq-js` के माध्यम से ब्राउज़र जैसा टीएलएस फ़िंगरप्रिंट --**🔏 सीएलआई फ़िंगरप्रिंट मिलान**- मूल सीएलआई बाइनरी हस्ताक्षरों से मिलान करने के लिए हेडर और बॉडी फ़ील्ड को पुन: व्यवस्थित करता है, जिससे खाता फ़्लैगिंग जोखिम काफी कम हो जाता है। प्रॉक्सी आईपी संरक्षित है - आपको एक साथ स्टील्थ**और**आईपी मास्किंग दोनों मिलते हैं
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. -<विवरण> -<सारांश>🆓 4. "मैं कोडिंग के लिए AI का उपयोग करना चाहता हूं लेकिन मेरे पास पैसे नहीं हैं" +**How OmniRoute solves it:** -हर कोई AI सदस्यता के लिए $20-200/माह का भुगतान नहीं कर सकता। छात्रों, उभरते देशों के डेवलपर्स, शौकीनों और फ्रीलांसरों को शून्य लागत पर गुणवत्ता वाले मॉडल तक पहुंच की आवश्यकता है। +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**ओम्नीरूट इसे कैसे हल करता है:** + --**Free Tier Providers Built-in**— Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, विज़न-मॉडल), किरो (क्लाउड + एडब्ल्यूएस बिल्डर आईडी मुफ़्त), जेमिनी सीएलआई (180K टोकन/माह मुफ़्त) --**ओलामा क्लाउड**- निःशुल्क "लाइट उपयोग" स्तर के साथ `api.ollama.com` पर क्लाउड-होस्टेड ओलामा मॉडल; `ollamacloud/` उपसर्ग का उपयोग करें --**केवल-नि:शुल्क कॉम्बो**- चेन `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/माह शून्य डाउनटाइम के साथ --**एनवीडिया एनआईएम फ्री एक्सेस**- ~40 आरपीएम डेव-बिल्ड.एनवीडिया.कॉम पर 70+ मॉडलों तक हमेशा के लिए मुफ्त एक्सेस (क्रेडिट से शुद्ध दर सीमा तक संक्रमण) --**लागत अनुकूलित रणनीति**- रूटिंग रणनीति जो स्वचालित रूप से सबसे सस्ते उपलब्ध प्रदाता को चुनती है +
+🆓 4. "I want to use AI for coding but I have no money" -<विवरण> -<सारांश>🔒 5. "मुझे अपने AI गेटवे को अनधिकृत पहुंच से बचाने की आवश्यकता है" +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -नेटवर्क (LAN, VPS, Docker) में AI गेटवे को उजागर करते समय, पते वाला कोई भी व्यक्ति डेवलपर के टोकन/कोटा का उपभोग कर सकता है। सुरक्षा के बिना, एपीआई दुरुपयोग, त्वरित इंजेक्शन और दुरुपयोग के प्रति संवेदनशील हैं। +**How OmniRoute solves it:** -**ओम्नीरूट इसे कैसे हल करता है:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**एपीआई कुंजी प्रबंधन**- एक समर्पित `/डैशबोर्ड/एपीआई-मैनेजर` पेज के साथ प्रति प्रदाता जेनरेशन, रोटेशन और स्कोपिंग --**मॉडल-स्तरीय अनुमतियाँ**- सभी को अनुमति दें/प्रतिबंधित टॉगल के साथ एपीआई कुंजियों को विशिष्ट मॉडल (`ओपनाई/*`, वाइल्डकार्ड पैटर्न) तक सीमित करें --**एपीआई एंडपॉइंट सुरक्षा**- `/v1/मॉडल` के लिए एक कुंजी की आवश्यकता है और लिस्टिंग से विशिष्ट प्रदाताओं को ब्लॉक करें --**ऑथ गार्ड + सीएसआरएफ सुरक्षा**- सभी डैशबोर्ड रूट `withAuth` मिडलवेयर + सीएसआरएफ टोकन से सुरक्षित हैं --**रेट लिमिटर**- कॉन्फ़िगर करने योग्य विंडो के साथ प्रति-आईपी दर सीमित करना --**आईपी फ़िल्टरिंग**- अभिगम नियंत्रण के लिए अनुमति सूची/अवरुद्ध सूची --**प्रॉम्प्ट इंजेक्शन गार्ड**- दुर्भावनापूर्ण प्रॉम्प्ट पैटर्न के विरुद्ध स्वच्छता --**एईएस-256-जीसीएम एन्क्रिप्शन**- क्रेडेंशियल आराम से एन्क्रिप्ट किए गए
+ -<विवरण> -<सारांश>🛑 6. "मेरा प्रदाता बंद हो गया और मैंने अपना कोडिंग प्रवाह खो दिया" +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -एआई प्रदाता अस्थिर हो सकते हैं, 5xx त्रुटियाँ लौटा सकते हैं, या अस्थायी दर सीमा तक पहुँच सकते हैं। यदि कोई डेवलपर किसी एकल प्रदाता पर निर्भर करता है, तो वे बाधित हो जाते हैं। सर्किट ब्रेकर के बिना, बार-बार पुनः प्रयास करने से एप्लिकेशन क्रैश हो सकता है। +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**ओम्नीरूट इसे कैसे हल करता है:** +**How OmniRoute solves it:** --**प्रति-मॉडल सर्किट ब्रेकर**- कॉन्फ़िगर करने योग्य थ्रेसहोल्ड और कूलडाउन (बंद/खुला/आधा-खुला) के साथ ऑटो-खुला/बंद, कैस्केडिंग ब्लॉक से बचने के लिए प्रति-मॉडल स्कोप्ड --**एक्सपोनेंशियल बैकऑफ़**- प्रगतिशील पुनः प्रयास में देरी --**एंटी-थंडरिंग हर्ड**- म्यूटेक्स + समवर्ती रिट्री तूफानों के खिलाफ सेमाफोर सुरक्षा --**कॉम्बो फ़ॉलबैक चेन**- यदि प्राथमिक प्रदाता विफल हो जाता है, तो बिना किसी हस्तक्षेप के स्वचालित रूप से चेन से गिर जाता है --**कॉम्बो सर्किट ब्रेकर**- कॉम्बो श्रृंखला के भीतर विफल प्रदाताओं को स्वचालित रूप से अक्षम करता है --**स्वास्थ्य डैशबोर्ड**- अपटाइम मॉनिटरिंग, सर्किट ब्रेकर स्थिति, लॉकआउट, कैश आँकड़े, p50/p95/p99 विलंबता
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest -<विवरण> -<सारांश>🔧 7. "प्रत्येक AI उपकरण को कॉन्फ़िगर करना कठिन और दोहराव वाला है" + -डेवलपर्स कर्सर, क्लाउड कोड, कोडेक्स सीएलआई, ओपनक्लाव, जेमिनी सीएलआई, किलो कोड का उपयोग करते हैं... प्रत्येक टूल को एक अलग कॉन्फ़िगरेशन (एपीआई एंडपॉइंट, कुंजी, मॉडल) की आवश्यकता होती है। प्रदाताओं या मॉडलों को स्विच करते समय पुन: कॉन्फ़िगर करना समय की बर्बादी है। +
+🛑 6. "My provider went down and I lost my coding flow" -**ओम्नीरूट इसे कैसे हल करता है:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**सीएलआई टूल्स डैशबोर्ड**- क्लाउड कोड, कोडेक्स सीएलआई, ओपनक्लाव, किलो कोड, एंटीग्रेविटी, क्लाइन के लिए एक-क्लिक सेटअप वाला समर्पित पेज --**गिटहब कोपायलट कॉन्फिग जेनरेटर**- बल्क मॉडल चयन के साथ वीएस कोड के लिए `चैटलैंग्वेजमॉडल.जेसन` जेनरेट करता है --**ऑनबोर्डिंग विज़ार्ड**- पहली बार उपयोगकर्ताओं के लिए निर्देशित 4-चरणीय सेटअप --**एक समापन बिंदु, सभी मॉडल**- `http://localhost:20128/v1` को एक बार कॉन्फ़िगर करें, 60+ प्रदाताओं तक पहुंचें
+**How OmniRoute solves it:** -<विवरण> -<सारांश>🔑 8. "एकाधिक प्रदाताओं से OAuth टोकन प्रबंधित करना नरक है" +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -क्लाउड कोड, कोडेक्स, जेमिनी सीएलआई, कोपायलट - सभी समाप्त होने वाले टोकन के साथ OAuth 2.0 का उपयोग करते हैं। डेवलपर्स को लगातार पुन: प्रमाणित करने, `client_secret is missing`, `redirect_uri_mismatch` और दूरस्थ सर्वर पर विफलताओं से निपटने की आवश्यकता होती है। LAN/VPS पर OAuth विशेष रूप से समस्याग्रस्त है। + -**ओम्नीरूट इसे कैसे हल करता है:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**ऑटो टोकन रिफ्रेश**- OAuth टोकन समाप्ति से पहले पृष्ठभूमि में रिफ्रेश होते हैं --**OAuth 2.0 (PKCE) बिल्ट-इन**- क्लाउड कोड, कोडेक्स, जेमिनी सीएलआई, कोपायलट, किरो, क्वेन, कोडर के लिए स्वचालित प्रवाह --**मल्टी-अकाउंट OAuth**- JWT/ID टोकन निष्कर्षण के माध्यम से प्रति प्रदाता एकाधिक खाते --**OAuth LAN/रिमोट फिक्स**- `redirect_uri` के लिए निजी आईपी पहचान + रिमोट सर्वर के लिए मैनुअल यूआरएल मोड --**Nginx के पीछे OAuth**- रिवर्स प्रॉक्सी संगतता के लिए `window.location.origin` का उपयोग करता है --**दूरस्थ OAuth मार्गदर्शिका**— VPS/Docker पर Google क्लाउड क्रेडेंशियल के लिए चरण-दर-चरण मार्गदर्शिका
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. -<विवरण> -<सारांश>📊 9. "मुझे नहीं पता कि मैं कितना और कहां खर्च कर रहा हूं" +**How OmniRoute solves it:** -डेवलपर्स कई भुगतान प्रदाताओं का उपयोग करते हैं लेकिन खर्च के बारे में कोई एकीकृत दृष्टिकोण नहीं रखते हैं। प्रत्येक प्रदाता का अपना बिलिंग डैशबोर्ड होता है, लेकिन कोई समेकित दृश्य नहीं होता है। अप्रत्याशित लागतें बढ़ सकती हैं। +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**ओम्नीरूट इसे कैसे हल करता है:** + --**लागत विश्लेषण डैशबोर्ड**— प्रति प्रदाता प्रति टोकन लागत ट्रैकिंग और बजट प्रबंधन --**प्रति स्तर बजट सीमा**- प्रति स्तर खर्च की अधिकतम सीमा जो स्वचालित फ़ॉलबैक को ट्रिगर करती है --**प्रति-मॉडल मूल्य निर्धारण कॉन्फ़िगरेशन**- प्रति मॉडल कॉन्फ़िगर करने योग्य कीमतें --**प्रति एपीआई कुंजी उपयोग सांख्यिकी**- अनुरोध गणना और प्रति कुंजी अंतिम बार उपयोग किया गया टाइमस्टैम्प --**एनालिटिक्स डैशबोर्ड**- स्टेट कार्ड, मॉडल उपयोग चार्ट, सफलता दर और विलंबता के साथ प्रदाता तालिका +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" -<विवरण> -<सारांश>🐛 10. "मैं एआई कॉल में त्रुटियों और समस्याओं का निदान नहीं कर सकता" +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -जब कोई कॉल विफल हो जाती है, तो देव को पता नहीं चलता कि यह दर सीमा, समाप्त टोकन, गलत प्रारूप या प्रदाता त्रुटि थी। विभिन्न टर्मिनलों पर खंडित लॉग। अवलोकन के बिना, डिबगिंग परीक्षण-और-त्रुटि है। +**How OmniRoute solves it:** -**ओम्नीरूट इसे कैसे हल करता है:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**एकीकृत लॉग डैशबोर्ड**- 4 टैब: अनुरोध लॉग, प्रॉक्सी लॉग, ऑडिट लॉग, कंसोल --**कंसोल लॉग व्यूअर**- रंग-कोडित स्तरों, ऑटो-स्क्रॉल, खोज, फ़िल्टर के साथ वास्तविक समय टर्मिनल-शैली व्यूअर --**SQLite प्रॉक्सी लॉग्स**- लगातार लॉग जो सर्वर पुनरारंभ होने से बचे रहते हैं --**अनुवादक खेल का मैदान**- 4 डिबगिंग मोड: खेल का मैदान (प्रारूप अनुवाद), चैट टेस्टर (राउंड-ट्रिप), टेस्ट बेंच (बैच), लाइव मॉनिटर (वास्तविक समय) --**अनुरोध टेलीमेट्री**- p50/p95/p99 विलंबता + X-अनुरोध-आईडी ट्रेसिंग --**रोटेशन के साथ फ़ाइल-आधारित लॉगिंग**- ऐप लॉग आकार, अवधारण दिनों और संग्रह गणना के अनुसार घूमते हैं; कॉल लॉग कलाकृतियाँ अवधारण दिनों और फ़ाइल गणना के अनुसार घूमती हैं --**सिस्टम जानकारी रिपोर्ट**- `npm run system-info` आपके पूर्ण वातावरण (नोड संस्करण, ओमनीरूट संस्करण, ओएस, सीएलआई उपकरण, डॉकर/पीएम2 स्थिति) के साथ `system-info.txt` उत्पन्न करता है। त्वरित ट्राइएज के लिए समस्याओं की रिपोर्ट करते समय इसे संलग्न करें।
+ -<विवरण> -<सारांश>🏗️ 11. "प्रवेश द्वार की तैनाती और रखरखाव जटिल है" +
+📊 9. "I don't know how much I'm spending or where" -विभिन्न वातावरणों (स्थानीय, वीपीएस, डॉकर, क्लाउड) में एआई प्रॉक्सी को स्थापित करना, कॉन्फ़िगर करना और बनाए रखना श्रम-गहन है। हार्डकोडेड पथ, निर्देशिकाओं पर `EACCES`, पोर्ट विरोध और क्रॉस-प्लेटफ़ॉर्म बिल्ड जैसी समस्याएं घर्षण बढ़ाती हैं। +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**ओम्नीरूट इसे कैसे हल करता है:** +**How OmniRoute solves it:** --**एनपीएम ग्लोबल इंस्टाल**- `एनपीएम इंस्टाल -जी ऑम्निरूटे && ऑम्निरूटे` - हो गया --**डॉकर मल्टी-प्लेटफ़ॉर्म**- AMD64 + ARM64 नेटिव (Apple सिलिकॉन, AWS ग्रेविटॉन, रास्पबेरी पाई) --**डॉकर कंपोज प्रोफाइल**- `बेस` (कोई सीएलआई उपकरण नहीं) और `सीएलआई` (क्लाउड कोड, कोडेक्स, ओपनक्लाव के साथ) --**इलेक्ट्रॉन डेस्कटॉप ऐप**- सिस्टम ट्रे, ऑटो-स्टार्ट, ऑफ़लाइन मोड के साथ विंडोज/मैकओएस/लिनक्स के लिए मूल ऐप --**स्प्लिट-पोर्ट मोड**- उन्नत परिदृश्यों के लिए अलग-अलग पोर्ट पर एपीआई और डैशबोर्ड (रिवर्स प्रॉक्सी, कंटेनर नेटवर्किंग) --**क्लाउड सिंक**- क्लाउडफ्लेयर वर्कर्स के माध्यम से सभी डिवाइसों में कॉन्फिग सिंक्रोनाइजेशन --**डीबी बैकअप**- बाह्य रूप से प्रबंधित बैकअप के लिए `DISABLE_SQLITE_AUTO_BACKUP` के साथ सभी सेटिंग्स का स्वचालित बैकअप, पुनर्स्थापना, निर्यात और आयात
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency -<विवरण> -<सारांश>🌍 12. "इंटरफ़ेस केवल अंग्रेजी है और मेरी टीम अंग्रेजी नहीं बोलती है" + -गैर-अंग्रेजी भाषी देशों, विशेष रूप से लैटिन अमेरिका, एशिया और यूरोप में टीमें, केवल अंग्रेजी इंटरफेस के साथ संघर्ष करती हैं। भाषा बाधाएँ अपनाने को कम करती हैं और कॉन्फ़िगरेशन त्रुटियों को बढ़ाती हैं। +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**ओम्नीरूट इसे कैसे हल करता है:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**डैशबोर्ड i18n - 30 भाषाएँ**- अरबी, बल्गेरियाई, डेनिश, जर्मन, स्पेनिश, फिनिश, फ्रेंच, हिब्रू, हिंदी, हंगेरियन, इंडोनेशियाई, इतालवी, जापानी, कोरियाई, मलय, डच, नॉर्वेजियन, पोलिश, पुर्तगाली (पीटी/बीआर), रोमानियाई, रूसी, स्लोवाक, स्वीडिश, थाई, यूक्रेनी, वियतनामी, चीनी, फिलिपिनो, अंग्रेजी सहित सभी 500+ कुंजियाँ अनुवादित --**आरटीएल समर्थन**- अरबी और हिब्रू के लिए दाएं से बाएं समर्थन --**बहु-भाषा रीडमी**- 30 पूर्ण दस्तावेज़ीकरण अनुवाद --**भाषा चयनकर्ता**- वास्तविक समय स्विचिंग के लिए हेडर में ग्लोब आइकन
+**How OmniRoute solves it:** -<विवरण> -<सारांश>🔄 13. "मुझे चैट से अधिक की आवश्यकता है - मुझे एम्बेडिंग, चित्र, ऑडियो की आवश्यकता है" +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -एआई का मतलब सिर्फ चैट पूरा करना नहीं है। डेवलपर्स को छवियां उत्पन्न करने, ऑडियो ट्रांसक्राइब करने, आरएजी के लिए एम्बेडिंग बनाने, दस्तावेज़ों को फिर से रैंक करने और सामग्री को मॉडरेट करने की आवश्यकता होती है। प्रत्येक एपीआई का एक अलग समापन बिंदु और प्रारूप होता है। + -**ओम्नीरूट इसे कैसे हल करता है:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**एंबेडिंग**- `/v1/एंबेडिंग` 6 प्रदाताओं और 9+ मॉडल के साथ --**छवि निर्माण**- 10 प्रदाताओं और 20+ मॉडलों के साथ `/v1/छवियां/पीढ़ी` (ओपनएआई, एक्सएआई, टुगेदर, फायरवर्क्स, नेबियस, हाइपरबोलिक, नैनोबनाना, एंटीग्रेविटी, एसडी वेबयूआई, कॉम्फीयूआई) --**टेक्स्ट-टू-वीडियो**- `/v1/वीडियो/पीढ़ी` - कॉम्फीयूआई (एनिमेटडिफ, एसवीडी) और एसडी वेबयूआई --**टेक्स्ट-टू-म्यूजिक**- `/v1/म्यूजिक/पीढ़ी` - कॉम्फीयूआई (स्थिर ऑडियो ओपन, म्यूजिकजेन) --**ऑडियो ट्रांसक्रिप्शन**- `/v1/ऑडियो/ट्रांसक्रिप्शन` - व्हिस्पर + एनवीडिया एनआईएम, हगिंगफेस, क्वेन3 --**टेक्स्ट-टू-स्पीच**- `/v1/ऑडियो/स्पीच` - इलेवनलैब्स, एनवीडिया एनआईएम, हगिंगफेस, कोक्वी, टोरटोइज़, क्वेन3,**इनवर्ल्ड**,**कार्टेसिया**,**प्लेएचटी**, + मौजूदा प्रदाता --**मॉडरेशन**- `/v1/मॉडरेशन` - सामग्री सुरक्षा जांच --**रीरैंकिंग**— `/v1/rerank` — दस्तावेज़ प्रासंगिकता रीरैंकिंग --**प्रतिक्रिया एपीआई**- कोडेक्स के लिए पूर्ण `/v1/प्रतिक्रिया` समर्थन
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. -<विवरण> -<सारांश>🧪 14. "मेरे पास सभी मॉडलों की गुणवत्ता का परीक्षण और तुलना करने का कोई तरीका नहीं है" +**How OmniRoute solves it:** -डेवलपर्स जानना चाहते हैं कि उनके उपयोग के मामले में कौन सा मॉडल सबसे अच्छा है - कोड, अनुवाद, तर्क - लेकिन मैन्युअल रूप से तुलना करना धीमा है। कोई एकीकृत eval उपकरण मौजूद नहीं है। +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**ओम्नीरूट इसे कैसे हल करता है:** + --**एलएलएम मूल्यांकन**- अभिवादन, गणित, भूगोल, कोड जनरेशन, JSON अनुपालन, अनुवाद, मार्कडाउन, सुरक्षा इनकार को कवर करने वाले 10 प्री-लोडेड मामलों के साथ गोल्डन सेट परीक्षण --**4 मिलान रणनीतियाँ**- `सटीक`, `शामिल`, `रेगेक्स`, `कस्टम` (जेएस फ़ंक्शन) --**अनुवादक खेल का मैदान परीक्षण बेंच**- एकाधिक इनपुट और अपेक्षित आउटपुट, क्रॉस-प्रदाता तुलना के साथ बैच परीक्षण --**चैट परीक्षक**- दृश्य प्रतिक्रिया प्रतिपादन के साथ पूर्ण राउंड-ट्रिप --**लाइव मॉनिटर**- प्रॉक्सी के माध्यम से बहने वाले सभी अनुरोधों की वास्तविक समय स्ट्रीम +
+🌍 12. "The interface is English-only and my team doesn't speak English" -<विवरण> -<सारांश>📈 15. "मुझे प्रदर्शन खोए बिना स्केल करने की आवश्यकता है" +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -जैसे-जैसे अनुरोध की मात्रा बढ़ती है, कैशिंग के बिना वही प्रश्न डुप्लिकेट लागत उत्पन्न करते हैं। निष्क्रियता के बिना, डुप्लिकेट अपशिष्ट प्रसंस्करण का अनुरोध करता है। प्रति-प्रदाता दर सीमा का सम्मान किया जाना चाहिए। +**How OmniRoute solves it:** -**ओम्नीरूट इसे कैसे हल करता है:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**सिमेंटिक कैश**- दो-स्तरीय कैश (हस्ताक्षर + सिमेंटिक) लागत और विलंबता को कम करता है --**अनुरोध Idempotency**- समान अनुरोधों के लिए 5s डिडुप्लीकेशन विंडो --**दर सीमा का पता लगाना**- प्रति-प्रदाता आरपीएम, न्यूनतम अंतर, और अधिकतम समवर्ती ट्रैकिंग --**संपादन योग्य दर सीमाएँ**— सेटिंग्स में कॉन्फ़िगर करने योग्य डिफ़ॉल्ट → दृढ़ता के साथ लचीलापन --**एपीआई कुंजी सत्यापन कैश**- उत्पादन प्रदर्शन के लिए 3-स्तरीय कैश --**टेलीमेट्री के साथ स्वास्थ्य डैशबोर्ड**— p50/p95/p99 विलंबता, कैश आँकड़े, अपटाइम
+ -<विवरण> -<सारांश>🤖 16. "मैं विश्व स्तर पर मॉडल व्यवहार को नियंत्रित करना चाहता हूं" +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -ऐसे डेवलपर जो सभी प्रतिक्रियाएं एक विशिष्ट भाषा में, एक विशिष्ट लहजे में चाहते हैं, या तर्क टोकन को सीमित करना चाहते हैं। प्रत्येक टूल/अनुरोध में इसे कॉन्फ़िगर करना अव्यावहारिक है। +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**ओम्नीरूट इसे कैसे हल करता है:** +**How OmniRoute solves it:** --**सिस्टम प्रॉम्प्ट इंजेक्शन**— ग्लोबल प्रॉम्प्ट सभी अनुरोधों पर लागू होता है --**सोच बजट सत्यापन**- प्रति अनुरोध तर्क टोकन आवंटन नियंत्रण (पासथ्रू, ऑटो, कस्टम, अनुकूली) --**9 रूटिंग रणनीतियाँ**- वैश्विक रणनीतियाँ जो यह निर्धारित करती हैं कि अनुरोध कैसे वितरित किए जाते हैं --**वाइल्डकार्ड राउटर**- `प्रदाता/*` पैटर्न किसी भी प्रदाता को गतिशील रूप से रूट करता है --**कॉम्बो सक्षम/अक्षम टॉगल**— कॉम्बो को सीधे डैशबोर्ड से टॉगल करें --**प्रदाता टॉगल**— एक क्लिक से प्रदाता के लिए सभी कनेक्शन सक्षम/अक्षम करें --**अवरुद्ध प्रदाता**- `/v1/मॉडल` सूची से विशिष्ट प्रदाताओं को बाहर निकालें
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex -<विवरण> -<सारांश>🧰 17. "मुझे प्रथम श्रेणी उत्पाद क्षमताओं के रूप में एमसीपी टूल्स की आवश्यकता है" + -कई एआई गेटवे एमसीपी को केवल एक छिपे हुए कार्यान्वयन विवरण के रूप में उजागर करते हैं। टीमों को एक दृश्यमान, प्रबंधनीय संचालन परत की आवश्यकता होती है। +
+🧪 14. "I have no way to test and compare quality across models" -**ओम्नीरूट इसे कैसे हल करता है:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- एमसीपी डैशबोर्ड नेविगेशन और एंडपॉइंट प्रोटोकॉल टैब में दिखाई देता है -- प्रक्रिया, उपकरण, कार्यक्षेत्र और ऑडिट के साथ समर्पित एमसीपी प्रबंधन पृष्ठ -- `omniroute --mcp` और क्लाइंट ऑनबोर्डिंग के लिए बिल्ट-इन क्विक-स्टार्ट
+**How OmniRoute solves it:** -<विवरण> -<सारांश>🧠 18. "मुझे सिंक + स्ट्रीम कार्य पथों के साथ A2A ऑर्केस्ट्रेशन की आवश्यकता है" +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -एजेंट वर्कफ़्लो को जीवनचक्र नियंत्रण के साथ सीधे उत्तर और लंबे समय तक चलने वाले स्ट्रीम निष्पादन दोनों की आवश्यकता होती है। + -**ओम्नीरूट इसे कैसे हल करता है:** +
+📈 15. "I need to scale without losing performance" -- A2A JSON-RPC एंडपॉइंट (`POST /a2a`) `मैसेज/सेंड` और `मैसेज/स्ट्रीम` के साथ -- टर्मिनल राज्य प्रसार के साथ एसएसई स्ट्रीमिंग -- `कार्य/प्राप्त करें` और `कार्य/रद्द करें` के लिए कार्य जीवनचक्र एपीआई
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. -<विवरण> -<सारांश>🛰️ 19. "मुझे वास्तविक एमसीपी प्रक्रिया स्वास्थ्य की आवश्यकता है, अनुमानित स्थिति की नहीं" +**How OmniRoute solves it:** -परिचालन टीमों को यह जानने की जरूरत है कि क्या एमसीपी वास्तव में जीवित है, न कि केवल एपीआई पहुंच योग्य है या नहीं। +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**ओम्नीरूट इसे कैसे हल करता है:** + -- पीआईडी, टाइमस्टैम्प, ट्रांसपोर्ट, टूल काउंट और स्कोप मोड के साथ रनटाइम हार्टबीट फ़ाइल -- एमसीपी स्थिति एपीआई दिल की धड़कन + हाल की गतिविधि का संयोजन -- प्रक्रिया/अपटाइम/दिल की धड़कन ताजगी के लिए यूआई स्टेटस कार्ड +
+🤖 16. "I want to control model behavior globally" -<विवरण> -<सारांश>📋 20. "मुझे ऑडिटेबल एमसीपी टूल निष्पादन की आवश्यकता है" +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -जब उपकरण कॉन्फ़िगरेशन को बदलते हैं या ऑप्स क्रियाओं को ट्रिगर करते हैं, तो टीमों को फोरेंसिक ट्रैसेबिलिटी की आवश्यकता होती है। +**How OmniRoute solves it:** -**ओम्नीरूट इसे कैसे हल करता है:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- MCP टूल कॉल के लिए SQLite समर्थित ऑडिट लॉगिंग -- टूल, सफलता/असफलता, एपीआई कुंजी और पेजिनेशन द्वारा फ़िल्टर -- डैशबोर्ड ऑडिट टेबल + स्वचालन के लिए आँकड़े समापन बिंदु
+ -<विवरण> -<सारांश>🔐 21. "मुझे प्रति एकीकरण के लिए स्कोप्ड एमसीपी अनुमतियों की आवश्यकता है" +
+🧰 17. "I need MCP tools as first-class product capabilities" -विभिन्न ग्राहकों को टूल श्रेणियों तक कम से कम विशेषाधिकार प्राप्त होना चाहिए। +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**ओम्नीरूट इसे कैसे हल करता है:** +**How OmniRoute solves it:** -- नियंत्रित टूल एक्सेस के लिए 10 दानेदार एमसीपी स्कोप -- एमसीपी प्रबंधन यूआई में दायरा प्रवर्तन और दृश्यता -- परिचालन टूलींग के लिए सुरक्षित डिफ़ॉल्ट मुद्रा
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding -<विवरण> -<सारांश>⚙️ 22. "मुझे पुनः तैनाती के बिना परिचालन नियंत्रण की आवश्यकता है" + -घटनाओं या लागत आयोजनों के दौरान टीमों को त्वरित रनटाइम परिवर्तन की आवश्यकता होती है। +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**ओम्नीरूट इसे कैसे हल करता है:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- कॉम्बो सक्रियण को सीधे एमसीपी डैशबोर्ड से स्विच करें -- पूर्व-निर्धारित पॉलिसी पैक से लचीलापन प्रोफ़ाइल लागू करें -- उसी ऑपरेशन पैनल से सर्किट ब्रेकर स्थिति को रीसेट करें
+**How OmniRoute solves it:** -<विवरण> -<सारांश>🔄 23. "मुझे लाइव ए2ए कार्य जीवनचक्र दृश्यता और रद्दीकरण की आवश्यकता है" +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -जीवनचक्र दृश्यता के बिना, कार्य घटनाओं का परीक्षण करना कठिन हो जाता है। + -**ओम्नीरूट इसे कैसे हल करता है:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- पेजिनेशन के साथ राज्य/कौशल द्वारा कार्य सूचीकरण/फ़िल्टरिंग -- कार्य मेटाडेटा, घटनाओं और कलाकृतियों पर ड्रिल-डाउन -- पुष्टि के साथ कार्य रद्दीकरण समापन बिंदु और यूआई कार्रवाई
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. -<विवरण> -<सारांश>🌊 24. "मुझे A2A लोड के लिए सक्रिय स्ट्रीम मेट्रिक्स की आवश्यकता है" +**How OmniRoute solves it:** -स्ट्रीमिंग वर्कफ़्लो के लिए समवर्ती और लाइव कनेक्शन में परिचालन अंतर्दृष्टि की आवश्यकता होती है। +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**ओम्नीरूट इसे कैसे हल करता है:** + -- सक्रिय स्ट्रीम काउंटर A2A स्थिति में एकीकृत -- अंतिम कार्य टाइमस्टैम्प और प्रति-राज्य गणना -- वास्तविक समय ऑप्स निगरानी के लिए A2A डैशबोर्ड कार्ड +
+📋 20. "I need auditable MCP tool execution" -<विवरण> -<सारांश>🪪 25. "मुझे ग्राहकों के लिए मानक एजेंट खोज की आवश्यकता है" +When tools mutate config or trigger ops actions, teams need forensic traceability. -बाहरी ग्राहकों और ऑर्केस्ट्रेटर्स को ऑनबोर्डिंग के लिए मशीन-पठनीय मेटाडेटा की आवश्यकता होती है। +**How OmniRoute solves it:** -**ओम्नीरूट इसे कैसे हल करता है:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- एजेंट कार्ड `/.well-known/agent.json` पर प्रदर्शित किया गया -- प्रबंधन यूआई में दिखाई गई क्षमताएं और कौशल -- A2A स्थिति API में स्वचालन के लिए खोज मेटाडेटा शामिल है
+ -<विवरण> +
+🔐 21. "I need scoped MCP permissions per integration" + +Different clients should have least-privilege access to tool categories. + +**How OmniRoute solves it:** + +- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling + +
+ +
+⚙️ 22. "I need operational controls without redeploying" + +Teams need quick runtime changes during incidents or cost events. + +**How OmniRoute solves it:** + +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel + +
+ +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" + +Without lifecycle visibility, task incidents become hard to triage. + +**How OmniRoute solves it:** + +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation + +
+ +
+🌊 24. "I need active stream metrics for A2A load" + +Streaming workflows require operational insight into concurrency and live connections. + +**How OmniRoute solves it:** + +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring + +
+ +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
🧭 26. "I need protocol discoverability in the product UX" -यदि उपयोगकर्ता प्रोटोकॉल सतहों की खोज नहीं कर पाते हैं, तो अपनाने और समर्थन की गुणवत्ता में गिरावट आती है। +If users cannot discover protocol surfaces, adoption and support quality drop. -**ओम्नीरूट इसे कैसे हल करता है:** +**How OmniRoute solves it:** -- प्रॉक्सी, एमसीपी, ए2ए और एपीआई एंडपॉइंट के लिए टैब के साथ समेकित**एंडपॉइंट**पेज -- एमसीपी और ए2ए के लिए इनलाइन सेवा स्थिति टॉगल (ऑनलाइन/ऑफ़लाइन)। -- सिंहावलोकन से लेकर समर्पित प्रबंधन टैब तक के लिंक
+- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs -<विवरण> -<सारांश>🧪 27. "मुझे वास्तविक ग्राहकों के साथ एंड-टू-एंड प्रोटोकॉल सत्यापन की आवश्यकता है" + -रिलीज़ से पहले प्रोटोकॉल संगतता को सत्यापित करने के लिए मॉक परीक्षण पर्याप्त नहीं हैं। +
+🧪 27. "I need end-to-end protocol validation with real clients" -**ओम्नीरूट इसे कैसे हल करता है:** +Mock tests are not enough to validate protocol compatibility before release. -- E2E सुइट जो ऐप को बूट करता है और वास्तविक MCP SDK क्लाइंट ट्रांसपोर्ट का उपयोग करता है -- A2A क्लाइंट खोज, भेजने, स्ट्रीम करने, प्राप्त करने और प्रवाह को रद्द करने के लिए परीक्षण करता है -- एमसीपी ऑडिट और ए2ए कार्य एपीआई के खिलाफ दावों की क्रॉस-चेक करें
+**How OmniRoute solves it:** -<विवरण> -<सारांश>📡 28. "मुझे सभी इंटरफेस में एकीकृत अवलोकन की आवश्यकता है" +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs -प्रोटोकॉल द्वारा अवलोकनशीलता को विभाजित करने से ब्लाइंड स्पॉट और लंबा एमटीटीआर बनता है। + -**ओम्नीरूट इसे कैसे हल करता है:** +
+📡 28. "I need unified observability across all interfaces" -- एक उत्पाद में एकीकृत डैशबोर्ड/लॉग/एनालिटिक्स -- स्वास्थ्य + ऑडिट + ओपनएआई, एमसीपी और ए2ए परतों में टेलीमेट्री अनुरोध -- स्थिति और स्वचालन के लिए परिचालन एपीआई
+Splitting observability by protocol creates blind spots and longer MTTR. -<विवरण> -<सारांश>💼 29. "मुझे प्रॉक्सी + टूल्स + एजेंट ऑर्केस्ट्रेशन के लिए एक रनटाइम की आवश्यकता है" +**How OmniRoute solves it:** -कई अलग-अलग सेवाएँ चलाने से परिचालन लागत और विफलता मोड बढ़ जाते हैं। +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation -**ओम्नीरूट इसे कैसे हल करता है:** + -- OpenAI-संगत प्रॉक्सी, MCP सर्वर और A2A सर्वर एक स्टैक में -- साझा प्रमाणीकरण, लचीलापन, डेटा भंडारण और अवलोकन क्षमता -- सभी संपर्क सतहों पर सुसंगत नीति मॉडल +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" -<विवरण> -<सारांश>🚀 30. "मुझे ग्लू-कोड फैलाव के बिना एजेंटिक वर्कफ़्लो भेजने की आवश्यकता है" +Running many separate services increases operational cost and failure modes. -कई तदर्थ सेवाओं और स्क्रिप्ट्स को सिलाई करते समय टीमों की गति कम हो जाती है। +**How OmniRoute solves it:** -**ओम्नीरूट इसे कैसे हल करता है:** +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces -- ग्राहकों और एजेंटों के लिए एकीकृत समापन बिंदु रणनीति -- अंतर्निहित प्रोटोकॉल प्रबंधन यूआई और धूम्रपान सत्यापन पथ -- उत्पादन के लिए तैयार नींव (सुरक्षा, लॉगिंग, लचीलापन, बैकअप)
+ + +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**प्लेबुक ए: सशुल्क सदस्यता + सस्ता बैकअप अधिकतम करें**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -607,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**प्लेबुक बी: शून्य-लागत कोडिंग स्टैक**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**प्लेबुक सी: 24/7 हमेशा चालू फ़ॉलबैक श्रृंखला**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -630,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**प्लेबुक डी: एजेंट एमसीपी + ए2ए के साथ काम करता है**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost ->**$0/माह**पर मिनटों में AI कोडिंग सेटअप करें। इन मुफ़्त खातों को कनेक्ट करें और अंतर्निहित**फ़्री स्टैक**कॉम्बो का उपयोग करें। +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| कदम | कार्रवाई | प्रदाता अनलॉक | +| Step | Action | Providers Unlocked | | ---- | -------------------------------------------------- | ------------------------------------------------------------------ | -| 1 | कनेक्ट**किरो**(AWS बिल्डर आईडी OAuth) | क्लाउड सॉनेट 4.5, हाइकु 4.5 -**असीमित**| -| 2 | कनेक्ट करें**Qoder**(Google OAuth) | किमी-के2-सोच, क्वेन3-कोडर-प्लस, डीपसीक-आर1... —**असीमित**| -| 3 | कनेक्ट**क्वेन**(डिवाइस कोड) | qwen3-कोडर-प्लस, qwen3-कोडर-फ़्लैश... —**असीमित**| -| 4 | कनेक्ट**मिथुन सीएलआई**(Google OAuth) | जेमिनी-3-फ़्लैश, जेमिनी-2.5-प्रो —**180K/महीना मुफ़्त**| -| 5 | `/डैशबोर्ड/कॉम्बोस` →**फ्री स्टैक ($0)**टेम्पलेट | सभी मुफ़्त प्रदाताओं को स्वचालित रूप से राउंड-रॉबिन करें | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**किसी भी आईडीई/सीएलआई को यहां इंगित करें:**`http://localhost:20128/v1` · एपीआई कुंजी: `any-string` · हो गया। +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**वैकल्पिक अतिरिक्त कवरेज (निःशुल्क भी):**ग्रोक एपीआई कुंजी (30 आरपीएम मुफ्त), एनवीडिया एनआईएम (40 आरपीएम मुफ्त, 70+ मॉडल), सेरेब्रस (1एम टोकन/दिन), लॉन्गकैट एपीआई कुंजी (50एम टोकन/दिन!), क्लाउडफ्लेयर वर्कर्स एआई (10के न्यूरॉन्स/दिन, 50+ मॉडल)।## त्वरित प्रारंभ +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## त्वरित प्रारंभ ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **pnpm उपयोगकर्ता:**`better-sqlite3` और `@swc/core` के लिए आवश्यक मूल बिल्ड स्क्रिप्ट को सक्षम करने के लिए इंस्टॉल के बाद `pnpm Approve-builds -g` चलाएँ: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > -> ```बैश -> पीएनपीएम इंस्टाल -जी ऑम्नीरूट -> पीएनपीएम अप्रूव-बिल्ड्स -जी # सभी पैकेजों का चयन करें → स्वीकृत करें -> सर्वमार्ग +> ```bash +> pnpm install -g omniroute +> pnpm approve-builds -g # Select all packages → approve +> omniroute > ``` -डैशबोर्ड `http://localhost:20128` पर खुलता है और एपीआई बेस यूआरएल `http://localhost:20128/v1` है। +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| आदेश | विवरण | -| ----------------------- | -------------------------------------------------------------------- | ----------- | -| `सर्वव्यापी` | सर्वर प्रारंभ करें (`पोर्ट=20128`, एपीआई और डैशबोर्ड एक ही पोर्ट पर) | -| `ओम्नीरूटे--पोर्ट 3000` | कैनोनिकल/एपीआई पोर्ट को 3000 | पर सेट करें | -| `omniroute --mcp` | MCP सर्वर (stdio ट्रांसपोर्ट) प्रारंभ करें | -| `omniroute --no-open` | ब्राउज़र को स्वतः न खोलें | -| `omniroute --help` | सहायता दिखाएँ | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -वैकल्पिक स्प्लिट-पोर्ट मोड:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -अधिकांश तैनाती के लिए, आपको केवल इसकी आवश्यकता है: +For most deployments, you only need: -| परिवर्तनीय | डिफ़ॉल्ट | उद्देश्य | -| ---------------------- | -------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600000` | अपस्ट्रीम फ़ेच, छिपे हुए अंडरसी टाइमआउट, टीएलएस फ़िंगरप्रिंट अनुरोध और एपीआई ब्रिज अनुरोध/प्रॉक्सी टाइमआउट के लिए साझा आधार रेखा | -| `STREAM_IDLE_TIMEOUT_MS` | `REQUEST_TIMEOUT_MS` प्राप्त होता है | ओम्नीरूट द्वारा एसएसई स्ट्रीम को निरस्त करने से पहले स्ट्रीमिंग खंडों के बीच अधिकतम अंतर +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -बैकवर्ड संगतता संरक्षित है: मौजूदा `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, और अन्य प्रति-लेयर टाइमआउट संस्करण अभी भी काम करते हैं और साझा बेसलाइन को ओवरराइड करते हैं। +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -यदि आपको बेहतर नियंत्रण की आवश्यकता है तो उन्नत ओवरराइड उपलब्ध हैं:| परिवर्तनीय | डिफ़ॉल्ट | उद्देश्य | -| ------------------------------------------------ | ------------------------------------------------ | ---------------------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | `REQUEST_TIMEOUT_MS` प्राप्त होता है | मुख्य फ़ेच एबॉर्ट सिग्नल द्वारा उपयोग किया गया कुल अपस्ट्रीम अनुरोध टाइमआउट | -| `FETCH_HEADERS_TIMEOUT_MS` | `FETCH_TIMEOUT_MS` विरासत में मिला है | अपस्ट्रीम प्रतिक्रिया हेडर प्राप्त करने के लिए निर्धारित समय सीमा | -| `FETCH_BODY_TIMEOUT_MS` | `FETCH_TIMEOUT_MS` विरासत में मिला है | अपस्ट्रीम बॉडी चंक्स के बीच निर्धारित समय सीमा (`0` इसे अक्षम कर देती है) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | अंडरसी टीसीपी कनेक्ट टाइमआउट | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | अन्य आइडल कीप-अलाइव सॉकेट टाइमआउट | -| `TLS_CLIENT_TIMEOUT_MS` | `FETCH_TIMEOUT_MS` विरासत में मिला है | `wreq-js` | के माध्यम से किए गए टीएलएस फिंगरप्रिंट अनुरोधों के लिए टाइमआउट -| `API_BRIDGE_PROXY_TIMEOUT_MS` | `REQUEST_TIMEOUT_MS` या `30000` प्राप्त होता है | एपीआई पोर्ट से डैशबोर्ड पोर्ट तक `/v1` प्रॉक्सी अग्रेषण के लिए टाइमआउट | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `अधिकतम(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | एपीआई ब्रिज सर्वर पर आने वाले अनुरोध का समय समाप्त | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | एपीआई ब्रिज सर्वर पर इनकमिंग हेडर टाइमआउट | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | एपीआई ब्रिज सर्वर पर कीप-अलाइव टाइमआउट | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | एपीआई ब्रिज सर्वर पर सॉकेट निष्क्रियता टाइमआउट (`0` इसे अक्षम करता है) | +Advanced overrides are available if you need finer control: -यदि आप Nginx, Caddy, Cloudflare, या किसी अन्य रिवर्स प्रॉक्सी के पीछे ओमनीरूट चलाते हैं, तो सुनिश्चित करें कि प्रॉक्सी -टाइमआउट आपके ओमनीरूट स्ट्रीम/फ़ेच टाइमआउट से भी अधिक हैं।### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. डैशबोर्ड → `प्रदाता` खोलें और कम से कम एक प्रदाता (OAuth या API कुंजी) कनेक्ट करें। -2. डैशबोर्ड → `एंडपॉइंट्स` खोलें और एक एपीआई कुंजी बनाएं। -3. (वैकल्पिक) डैशबोर्ड → `कॉम्बोस` खोलें और अपनी फ़ॉलबैक श्रृंखला सेट करें।### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -क्लाउड कोड, कोडेक्स सीएलआई, जेमिनी सीएलआई, कर्सर, क्लाइन, ओपनक्लाव, ओपनकोड और ओपनएआई-संगत एसडीके के साथ काम करता है।### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**एमसीपी (टूल-संचालित संचालन के लिए):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -फिर अपने एमसीपी क्लाइंट को `stdio` पर कनेक्ट करें और टूल का परीक्षण करें जैसे: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (एजेंट-टू-एजेंट वर्कफ़्लो के लिए):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -759,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -यह सुइट चल रहे ऐप के विरुद्ध वास्तविक MCP और A2A क्लाइंट प्रवाह को मान्य करता है।### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -767,13 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` -<विवरण> -<सारांश>शून्य लिनक्स (`xbps-src` टेम्पलेट) +
+Void Linux (`xbps-src` template) -Void Linux उपयोगकर्ताओं के लिए, आप `xbps-src` का उपयोग करके एक मूल पैकेज बना सकते हैं। इस ब्लॉक को `srcpkgs/omniroute/template` के रूप में सहेजें:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -785,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -793,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -869,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -880,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -ओमनीरूट [डॉकर हब](https://hub.docker.com/r/diegosouzapw/omniroute) पर सार्वजनिक डॉकर छवि के रूप में उपलब्ध है। +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**तेज़ भागना:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -890,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**पर्यावरण फ़ाइल के साथ:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**डॉकर कंपोज़ का उपयोग करना:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -डॉकर परिनियोजन के लिए डैशबोर्ड समर्थन में अब `डैशबोर्ड → एंडपॉइंट्स` पर एक-क्लिक**क्लाउडफ्लेयर क्विक टनल**शामिल है। पहला, जरूरत पड़ने पर ही `क्लाउडफ्लेयर` को डाउनलोड करने में सक्षम बनाता है, आपके वर्तमान `/v1` समापन बिंदु पर एक अस्थायी सुरंग शुरू करता है, और उत्पन्न `https://*.trycloudflare.com/v1` यूआरएल को सीधे आपके सामान्य सार्वजनिक यूआरएल के नीचे दिखाता है। +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -टिप्पणियाँ: +Notes: -- क्विक टनल यूआरएल अस्थायी होते हैं और हर पुनरारंभ के बाद बदल जाते हैं। -- ओम्निरूट या कंटेनर पुनरारंभ के बाद त्वरित सुरंगें स्वतः बहाल नहीं होती हैं। आवश्यकता पड़ने पर उन्हें डैशबोर्ड से पुनः सक्षम करें। -- प्रबंधित इंस्टाल वर्तमान में `x64` / `arm64` पर Linux, macOS और Windows का समर्थन करता है। -- प्रबंधित त्वरित सुरंगें प्रतिबंधित कंटेनर वातावरण में शोर वाले क्विक यूडीपी बफर चेतावनियों से बचने के लिए HTTP/2 परिवहन के लिए डिफ़ॉल्ट हैं। यदि आप एक अलग परिवहन चाहते हैं तो `CLOUDFLARED_PROTOCOL=quic` या `auto` सेट करें। -- डॉकर छवियां सिस्टम सीए रूट्स को बंडल करती हैं और उन्हें प्रबंधित `क्लाउडफ्लेयर` में भेजती हैं, जो कंटेनर के अंदर सुरंग बूटस्ट्रैप होने पर टीएलएस ट्रस्ट विफलताओं से बचाती है। -- SQLite वाल मोड में चलता है। `डॉकर स्टॉप` को समाप्त होने की अनुमति दी जानी चाहिए ताकि ओमनीरूट नवीनतम परिवर्तनों को `स्टोरेज.स्क्लाइट` में वापस चेकपॉइंट कर सके। -- बंडल की गई कंपोज़ फ़ाइलें पहले से ही 40s स्टॉप ग्रेस अवधि निर्धारित करती हैं। यदि आप छवि को सीधे चलाते हैं, तो `--स्टॉप-टाइमआउट 40` (या समान) रखें ताकि मैन्युअल स्टॉप शटडाउन क्लीनअप में कटौती न करें। -- यदि आप चाहते हैं कि ओमनीरूट डाउनलोड करने के बजाय मौजूदा बाइनरी का उपयोग करे तो `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` सेट करें। +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**कैडी (HTTPS ऑटो-टीएलएस) के साथ डॉकर कंपोज़ का उपयोग करना:** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -कैडी के स्वचालित एसएसएल प्रावधान का उपयोग करके ओमनीरूट को सुरक्षित रूप से उजागर किया जा सकता है। सुनिश्चित करें कि आपके डोमेन का DNS A रिकॉर्ड आपके सर्वर के आईपी को इंगित करता है।```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` +| Image | Tag | Size | Description | +| ------------------------ | -------- | ------ | --------------------- | +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | -| छवि | टैग | आकार | विवरण | -| ---------------------- | -------- | ------ | ---------------------- | -| `diegosouzapw/omniroute` | `नवीनतम` | ~250एमबी | नवीनतम स्थिर रिलीज़ | -| `diegosouzapw/omniroute` | `1.0.3` | ~250एमबी | वर्तमान संस्करण |--- +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**नया!**ओमनीरूट अब विंडोज, मैकओएस और लिनक्स के लिए**नेटिव डेस्कटॉप एप्लिकेशन**के रूप में उपलब्ध है। +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -ओम्नीरूट को एक स्टैंडअलोन डेस्कटॉप ऐप के रूप में चलाएं - स्थानीय मॉडलों के लिए कोई टर्मिनल, कोई ब्राउज़र, कोई इंटरनेट आवश्यक नहीं है। इलेक्ट्रॉन-आधारित ऐप में शामिल हैं: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**नेटिव विंडो**- सिस्टम ट्रे एकीकरण के साथ समर्पित ऐप विंडो -- 🔄**ऑटो-स्टार्ट**— सिस्टम लॉगिन पर ओमनीरूट लॉन्च करें -- 🔔**मूल सूचनाएं**- कोटा समाप्त होने या प्रदाता समस्याओं के लिए अलर्ट प्राप्त करें -- ⚡**वन-क्लिक इंस्टॉल**- एनएसआईएस (विंडोज़), डीएमजी (मैकओएस), ऐपइमेज (लिनक्स) -- 🌐**ऑफ़लाइन मोड**— बंडल सर्वर के साथ पूरी तरह ऑफ़लाइन काम करता है### त्वरित प्रारंभ +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### त्वरित प्रारंभ ```bash # Development mode @@ -979,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -न्यूनतम होने पर, ओमनीरूट त्वरित क्रियाओं के साथ आपके सिस्टम ट्रे में रहता है: +When minimized, OmniRoute lives in your system tray with quick actions: -- डैशबोर्ड खोलें -- सर्वर पोर्ट बदलें -- आवेदन छोड़ें +- Open dashboard +- Change server port +- Quit application -📖 पूर्ण दस्तावेज: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| टियर | प्रदाता | लागत | कोटा रीसेट | के लिए सर्वश्रेष्ठ | -| ----------------- | ---------------------------- | -------------------------------- | ---------------------- | -------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 सदस्यता** | क्लाउड कोड (प्रो) | $20/माह | 5 घंटे + साप्ताहिक | पहले ही सदस्यता ले ली है | -| | कोडेक्स (प्लस/प्रो) | $20-200/महीना | 5 घंटे + साप्ताहिक | OpenAI उपयोगकर्ता | -| | जेमिनी सीएलआई | **मुफ़्त** | 180K/माह + 1K/दिन | सब लोग! | -| | गिटहब कोपायलट | $10-19/माह | मासिक | GitHub उपयोगकर्ता | -| **🔑एपीआई कुंजी** | एनवीडिया एनआईएम | **मुफ़्त**(हमेशा के लिए देव) | ~40 आरपीएम | 70+ खुले मॉडल | -| | सेरेब्रस | **मुफ़्त**(1 मिलियन टोकन/दिन) | 60K टीपीएम / 30 आरपीएम | दुनिया का सबसे तेज़ | -| | ग्रोक | **मुफ़्त**(30 आरपीएम) | 14.4K आरपीडी | अल्ट्रा-फास्ट लामा/जेम्मा | -| | डीपसीक V3.2 | $0.27/$1.10 प्रति 1 मिलियन | कोई नहीं | सर्वोत्तम मूल्य/गुणवत्ता तर्क | -| | xAI ग्रोक-4 फास्ट | **$0.20/$0.50 प्रति 1 मिलियन**🆕 | कोई नहीं | सबसे तेज़ + टूल कॉलिंग, अल्ट्रालो | -| | xAI ग्रोक-4 (मानक) | $0.20/$1.50 प्रति 1 मिलियन 🆕 | कोई नहीं | xAI से रीज़निंग फ्लैगशिप | -| | मिस्ट्रल | नि:शुल्क परीक्षण + सशुल्क | दर सीमित | यूरोपीय एआई | -| | ओपनराउटर | भुगतान-प्रति-उपयोग | कोई नहीं | 100+ मॉडल कुल मिलाकर। | -| **💰सस्ता** | GLM-5 (Z.AI के माध्यम से) 🆕 | $0.5/1 मिलियन | प्रतिदिन सुबह 10 बजे | 128K आउटपुट, नवीनतम फ्लैगशिप | -| | जीएलएम-4.7 | $0.6/1 मिलियन | प्रतिदिन सुबह 10 बजे | बजट बैकअप | -| | मिनीमैक्स एम2.5 🆕 | $0.3/1M इनपुट | 5 घंटे की रोलिंग | तर्क + एजेंटिक कार्य | -| | मिनीमैक्स एम2.1 | $0.2/1 मिलियन | 5 घंटे की रोलिंग | सबसे सस्ता विकल्प | -| | किमी K2.5 (मूनशॉट एपीआई) 🆕 | भुगतान-प्रति-उपयोग | कोई नहीं | डायरेक्ट मूनशॉट एपीआई एक्सेस | -| | किमी K2 | $9/महीना फ्लैट | 10एम टोकन/माह | अनुमानित लागत | -| **🆓 मुफ़्त** | कोडर | **$0** | असीमित | 5 मॉडल असीमित | -| | क्वेन | **$0** | असीमित | 4 मॉडल असीमित | -| | किरो | **$0** | असीमित | क्लाउड सॉनेट/हाइकू (एडब्ल्यूएस बिल्डर) | -| | लॉन्गकैट फ्लैश-लाइट 🆕 | **$0**(50 मिलियन टोकन/दिन 🔥) | 1 आरपीएस | पृथ्वी पर सबसे बड़ा मुफ़्त कोटा | -| | परागण एआई 🆕 | **$0**(कोई कुंजी आवश्यक नहीं) | 1 अनुरोध/15s | जीपीटी-5, क्लाउड, डीपसीक, लामा 4 | -| | क्लाउडफ्लेयर वर्कर्स एआई 🆕 | **$0**(10K न्यूरॉन्स/दिन) | ~150 सम्मान/दिन | 50+ मॉडल, वैश्विक बढ़त | -| | स्केलवे एआई 🆕 | **$0**(कुल 1 मिलियन टोकन) | दर सीमित | ईयू/जीडीपीआर, क्वेन3 235बी, लामा 70बी | > 🆕**नए मॉडल जोड़े गए (मार्च 2026):**$0.20/$0.50/M पर ग्रोक-4 फास्ट परिवार (1143ms पर बेंचमार्क - जेमिनी 2.5 फ्लैश से 30% तेज), 128K आउटपुट के साथ Z.AI के माध्यम से GLM-5, मिनीमैक्स M2.5 रीजनिंग, डीपसीक V3.2 अद्यतन मूल्य निर्धारण, मूनशॉट डायरेक्ट एपीआई के माध्यम से किमी K2.5। | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 $0 कॉम्बो स्टैक - पूर्ण निःशुल्क सेटअप:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**शून्य लागत. कोडिंग कभी बंद नहीं होती।**इसे एक ओमनीरूट कॉम्बो के रूप में कॉन्फ़िगर करें और सभी फ़ॉलबैक स्वचालित रूप से होते हैं - कोई मैन्युअल स्विचिंग नहीं।--- +--- --- ## 🆓 Free Models — What You Actually Get -> नीचे दिए गए सभी मॉडल**शून्य क्रेडिट कार्ड की आवश्यकता के साथ 100% निःशुल्क**हैं। जब एक कोटा समाप्त हो जाता है तो ओमनीरूट उनके बीच ऑटो-रूट करता है - उन सभी को एक अटूट $0 कॉम्बो के लिए संयोजित करें।### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| मॉडल | उपसर्ग | सीमा | दर सीमा | -| ------------------- | ------ | ----------------- | ---------------------- | -| `क्लाउड-सॉनेट-4.5` | `क्र/` |**असीमित**| कोई रिपोर्ट नहीं की गई दैनिक सीमा | -| `क्लाउड-हाइकु-4.5` | `क्र/` |**Unlimited**| कोई रिपोर्ट नहीं की गई दैनिक सीमा | -| `क्लाउड-ओपस-4.6` | `क्र/` |**असीमित**| किरो के माध्यम से नवीनतम रचना |### 🟢 QODER MODELS (Free PAT via qodercli) +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) -| मॉडल | उपसर्ग | सीमा | दर सीमा | -| ------------------ | ------ | ----------------- | --------------- | -| `किमी-के2-सोच` | `अगर/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं | -| `क्वेन3-कोडर-प्लस` | `अगर/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं | -| `डीपसीक-आर1` | `अगर/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं | -| `मिनीमैक्स-एम2.1` | `अगर/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं | -| `किमी-के2` | `अगर/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं | +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | --------------------- | +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -> अनुशंसित कनेक्शन विधि:**पर्सनल एक्सेस टोकन + `qodercli`**। ब्राउज़र OAuth है -> प्रयोगात्मक और डिफ़ॉल्ट रूप से अक्षम जब तक कि `QODER_OAUTH_*` पर्यावरण चर कॉन्फ़िगर नहीं किए जाते।### 🟡 QWEN MODELS (Device Code Auth) +### 🟢 QODER MODELS (Free PAT via qodercli) -| मॉडल | उपसर्ग | सीमा | दर सीमा | -| ------------------- | ------ | ----------------- | ------------------- | -| `क्वेन3-कोडर-प्लस` | `qw/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं | -| `क्वेन3-कोडर-फ़्लैश` | `qw/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं | -| `qwen3-कोडर-नेक्स्ट` | `qw/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं | -| `विज़न-मॉडल` | `qw/` |**असीमित**| मल्टीमॉडल (चित्र) |### 🟣 GEMINI CLI (Google OAuth) +| Model | Prefix | Limit | Rate Limit | +| ------------------ | ------ | ------------- | --------------- | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -| मॉडल | उपसर्ग | सीमा | Rate Limit | -| ---------------------- | ------ | -------------------------------- | ----------------- | -| `मिथुन-3-फ़्लैश-पूर्वावलोकन` | `जीसी/` |**180K टोकन/माह**+ 1K/दिन | मासिक रीसेट | -| `मिथुन-2.5-प्रो` | `जीसी/` | 180K/माह (साझा पूल) | उच्च गुणवत्ता |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| टियर | दैनिक सीमा | दर सीमा | नोट्स | -| ---------- | ----------- | ----------- | ---------------------------------------------------------------- | -| मुफ़्त (देव) | कोई टोकन सीमा नहीं |**~40 आरपीएम**| 70+ मॉडल; 2025 के मध्य में शुद्ध दर सीमा में परिवर्तन | +### 🟡 QWEN MODELS (Device Code Auth) -लोकप्रिय मुफ्त मॉडल: `मूनशोताई/किमी-के2.5` (किमी के2.5), `जेड-एआई/जीएलएम4.7` (जीएलएम 4.7), `डीपसीक-एआई/डीपसीक-वी3.2` (डीपसीक वी3.2), `एनवीडिया/ल्लामा-3.3-70बी-इंस्ट्रक्ट`, `डीपसीक/डीपसीक-आर1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | ------------------- | +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| टियर | दैनिक सीमा | दर सीमा | नोट्स | -| ---- | ----------------- | ---------------- | ------------------------------------------------ | -| मुफ़्त |**1 मिलियन टोकन/दिन**| 60K टीपीएम / 30 आरपीएम | दुनिया का सबसे तेज़ एलएलएम अनुमान; प्रतिदिन रीसेट होता है | +### 🟣 GEMINI CLI (Google OAuth) -निःशुल्क उपलब्ध: `लामा-3.3-70बी`, `लामा-3.1-8बी`, `डीपसीक-आर1-डिस्टिल-लामा-70बी`### 🔴 GROQ (Free API Key — console.groq.com) +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -| टियर | दैनिक सीमा | दर सीमा | नोट्स | -| ---- | ----------------- | ---------------- | ------------------------------------------------ | -| मुफ़्त |**14.4K आरपीडी**| प्रति मॉडल 30 आरपीएम | कोई क्रेडिट कार्ड नहीं; 429 सीमा पर, शुल्क नहीं | +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) -निःशुल्क उपलब्ध: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +| Tier | Daily Limit | Rate Limit | Notes | +| ---------- | ------------ | ----------- | ------------------------------------------------------ | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -| मॉडल | उपसर्ग | दैनिक निःशुल्क कोटा | नोट्स | -| -------------------------------- | ------ | ----------------- | ---------------------- | -| `लॉन्गकैट-फ्लैश-लाइट` | `एलसी/` |**50M टोकन**💥 | अब तक का सबसे बड़ा मुफ़्त कोटा | -| 'लॉन्गकैट-फ्लैश-चैट' | `एलसी/` | 500K टोकन | मल्टी-टर्न चैट | -| 'लॉन्गकैट-फ्लैश-थिंकिंग' | `एलसी/` | 500K टोकन | तर्क/सीओटी | -| `लॉन्गकैट-फ्लैश-थिंकिंग-2601` | `एलसी/` | 500K टोकन | जनवरी 2026 संस्करण | -| `लॉन्गकैट-फ्लैश-ओमनी-2603` | `एलसी/` | 500K टोकन | मल्टीमॉडल | +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -> सार्वजनिक बीटा में रहते हुए 100% निःशुल्क। ईमेल या फोन से [longcat.chat](https://longcat.chat) पर साइन अप करें। प्रतिदिन 00:00 UTC पर रीसेट होता है।### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -| मॉडल | उपसर्ग | दर सीमा | पीछे प्रदाता | +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | + +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` + +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ------------- | ---------------- | ----------------------------------------- | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | + +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` + +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 + +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | + +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. + +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| 'ओपनाई' | `पोल/` | 1 अनुरोध/15s | जीपीटी-5 | -| 'क्लाउड' | `पोल/` | 1 अनुरोध/15s | एंथ्रोपिक क्लाउड | -| 'मिथुन' | `पोल/` | 1 अनुरोध/15s | गूगल जेमिनी | -| 'डीपसीक' | `पोल/` | 1 अनुरोध/15s | डीपसीक वी3 | -| 'लामा' | `पोल/` | 1 अनुरोध/15s | मेटा लामा 4 स्काउट | -| 'मिस्ट्रल' | `पोल/` | 1 अनुरोध/15s | मिस्ट्रल एआई | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**शून्य घर्षण:**कोई साइनअप नहीं, कोई एपीआई कुंजी नहीं। खाली कुंजी फ़ील्ड के साथ परागण प्रदाता जोड़ें और यह तुरंत काम करता है।### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| टियर | दैनिक न्यूरॉन्स | समतुल्य उपयोग | नोट्स | -| ---- | ----------------- | ------------------------------------------------ | ---------------------- | -| मुफ़्त |**10,000**| ~150 एलएलएम सम्मान / 500s ऑडियो / 15K एंबेड | वैश्विक बढ़त, 50+ मॉडल | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 -लोकप्रिय मुफ़्त मॉडल: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (मुफ़्त ऑडियो!), `@cf/qwen/qwen2.5-coder-15b-instruct` +| Tier | Daily Neurons | Equivalent Usage | Notes | +| ---- | ------------- | --------------------------------------- | ----------------------- | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -> [dash.cloudflare.com](https://dash.cloudflare.com) से एपीआई टोकन + खाता आईडी की आवश्यकता है। प्रदाता सेटिंग्स में खाता आईडी संग्रहीत करें।### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -| टियर | मुफ़्त कोटा | स्थान | नोट्स | -| ---- | ----------------- | ----------- | -------------------------------------- | -| मुफ़्त |**1M टोकन**| 🇫🇷पेरिस, ईयू | सीमा के भीतर किसी क्रेडिट कार्ड की आवश्यकता नहीं | +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -निःशुल्क उपलब्ध: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `depseek-v3-0324` +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 -> ईयू/जीडीपीआर के अनुरूप। [console.scaleway.com](https://console.scaleway.com) पर एपीआई कुंजी प्राप्त करें। +| Tier | Free Quota | Location | Notes | +| ---- | ------------- | ------------ | ----------------------------------- | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | ->**💡 परम निःशुल्क स्टैक (11 प्रदाता, $0 हमेशा के लिए):** +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` + +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). + +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> किरो (केआर/) → क्लाउड सॉनेट/हाइकु अनलिमिटेड -> कोडर (यदि/) → किमी-के2-सोच, क्वेन3-कोडर-प्लस, डीपसीक-आर1 अनलिमिटेड -> लॉन्गकैट लाइट (एलसी/) → लॉन्गकैट-फ्लैश-लाइट - 50एम टोकन/दिन 🔥 -> परागण (पोल/) → जीपीटी-5, क्लाउड, डीपसीक, लामा 4 - किसी कुंजी की आवश्यकता नहीं -> क्वेन (qw/) → क्वेन3-कोडर मॉडल असीमित -> मिथुन (मिथुन /) → मिथुन 2.5 फ्लैश - 1,500 अनुरोध/दिन निःशुल्क -> क्लाउडफ्लेयर एआई (सीएफ/) → 50+ मॉडल - 10K न्यूरॉन्स/दिन -> स्केलवे (scw/) → Qwen3 235B, Llama 70B — 1M मुफ़्त टोकन (EU) -> ग्रोक (ग्रोक/) → लामा/जेम्मा - 14.4K अनुरोध/दिन अल्ट्रा-फास्ट -> एनवीडिया एनआईएम (एनवीडिया/) → 70+ खुले मॉडल - 40 आरपीएम हमेशा के लिए -> सेरेब्रस (सेरेब्रस/) → लामा/क्वेन दुनिया का सबसे तेज़ - 1 मिलियन टोकन/दिन -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` ->**$0**के लिए किसी भी ऑडियो/वीडियो को ट्रांसक्राइब करें - डीपग्राम $200 मुफ़्त, असेंबलीएआई $50 फ़ॉलबैक, ग्रोक व्हिस्पर असीमित आपातकालीन बैकअप के साथ अग्रणी है। +## 🎙️ Free Transcription Combo -| प्रदाता | मुफ़्त क्रेडिट | सर्वश्रेष्ठ मॉडल | दर सीमा | -| ----------------- | ---------------------- | ------------------------------------------------ | -------------------------------- | -| 🟢**दीपग्राम**|**$200 निःशुल्क**(साइनअप) | `नोवा-3` — सर्वोत्तम सटीकता, 30+ भाषाएँ | निःशुल्क क्रेडिट पर कोई आरपीएम सीमा नहीं | -| 🔵**असेंबलीएआई**|**$50 निःशुल्क**(साइनअप) | `यूनिवर्सल-3-प्रो` — अध्याय, भावना, पीआईआई | निःशुल्क क्रेडिट पर कोई आरपीएम सीमा नहीं | -| 🔴**ग्रोक**|**हमेशा के लिए मुफ़्त**| `व्हिस्पर-लार्ज-v3` - ओपनएआई व्हिस्पर | 30 आरपीएम (दर सीमित) | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. -**`/डैशबोर्ड/कॉम्बोस` में सुझाया गया कॉम्बो:**``` +| Provider | Free Credits | Best Model | Rate Limit | +| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | + +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -फिर `/डैशबोर्ड/मीडिया` →**ट्रांसक्रिप्शन**टैब में: कोई भी ऑडियो या वीडियो फ़ाइल अपलोड करें → अपना कॉम्बो एंडपॉइंट चुनें → समर्थित प्रारूपों में ट्रांसक्रिप्शन प्राप्त करें।## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -ओमनीरूट v2.0 को केवल एक रिले प्रॉक्सी नहीं, बल्कि एक ऑपरेशनल प्लेटफॉर्म के रूप में बनाया गया है।### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| फ़ीचर | यह क्या करता है | -| ---------------------------------- | -------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**ग्रोक-4 फास्ट फ़ैमिली** | $0.20/$0.50/M पर xAI मॉडल - बेंचमार्क 1143ms (जेमिनी 2.5 फ्लैश से 30% तेज) | -| 🧠**Z.AI के माध्यम से GLM-5** | 128K आउटपुट संदर्भ, $0.5/1M - GLM परिवार का नवीनतम फ्लैगशिप | -| 🔮**मिनीमैक्स एम2.5** | $0.30/1 मिलियन पर रीज़निंग + एजेंटिक कार्य - एम2.1 से महत्वपूर्ण उन्नयन | -| 🎯**टूलकॉलिंग फ़्लैग प्रति मॉडल** | रजिस्ट्री में प्रति-मॉडल `टूलकॉलिंग: सही/गलत` - ऑटोकॉम्बो गैर-टूल-सक्षम मॉडल को छोड़ देता है | -| 🌍**बहुभाषी आशय का पता लगाना** | ऑटोकॉम्बो स्कोरिंग में पीटी/जेडएच/ईएस/एआर कीवर्ड - गैर-अंग्रेजी सामग्री के लिए बेहतर मॉडल चयन | -| 📊**बेंचमार्क-प्रेरित फ़ॉलबैक** | लाइव अनुरोधों से वास्तविक p95 विलंबता कॉम्बो स्कोरिंग फ़ीड करती है - ऑटोकॉम्बो वास्तविक डेटा से सीखता है | -| 🔁**डुप्लीकेशन का अनुरोध** | कंटेंट-हैश आधारित डिडअप विंडो - मल्टी-एजेंट सुरक्षित, डुप्लिकेट शुल्क को रोकता है | -| 🔌**प्लग करने योग्य राउटर रणनीति** | एक्स्टेंसिबल `राउटरस्ट्रैटेजी` इंटरफ़ेस - प्लगइन्स के रूप में कस्टम रूटिंग लॉजिक जोड़ें | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| फ़ीचर | यह क्या करता है | -| -------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------- | -| 🎮**मॉडल खेल का मैदान** | किसी भी मॉडल का सीधे परीक्षण करने के लिए डैशबोर्ड पेज - प्रदाता/मॉडल/एंडपॉइंट चयनकर्ता, मोनाको संपादक, स्ट्रीमिंग, निरस्त, समय | -| 🔏**सीएलआई फ़िंगरप्रिंट मिलान** | मूल सीएलआई हस्ताक्षरों से मिलान करने के लिए प्रति-प्रदाता हेडर/बॉडी ऑर्डरिंग - सेटिंग्स> सुरक्षा में प्रति प्रदाता टॉगल करें।**आपका प्रॉक्सी आईपी संरक्षित है** | -| 🤝**एसीपी सपोर्ट (एजेंट क्लाइंट प्रोटोकॉल)** | सीएलआई एजेंट खोज (कोडेक्स, क्लाउड, गूज़, जेमिनी सीएलआई, ओपनक्लॉ + 9 अधिक), प्रोसेस स्पॉनर, `/api/acp/agents` एंडपॉइंट | -| 🤖**एसीपी एजेंट्स डैशबोर्ड** | डीबग › एजेंट पेज - किसी भी सीएलआई टूल के लिए इंस्टॉल स्थिति, संस्करण, कस्टम एजेंट फॉर्म के साथ 14 एजेंटों का ग्रिड।**ओपनकोड**उपयोगकर्ताओं को एक "डाउनलोड ओपनकोड.जेसन" बटन मिलता है जो सभी उपलब्ध मॉडलों के साथ उपयोग के लिए तैयार कॉन्फ़िगरेशन को स्वतः उत्पन्न करता है। | -| 🔧**कस्टम मॉडल `एपीआईफॉर्मेट` रूटिंग** | `apiFormat: "प्रतिक्रियाएं"` के साथ कस्टम मॉडल अब प्रतिक्रिया एपीआई अनुवादक पर सही ढंग से रूट करते हैं | -| 🏢**कोडेक्स कार्यक्षेत्र अलगाव** | प्रति ईमेल एकाधिक कोडेक्स कार्यस्थान - OAuth कार्यस्थान आईडी द्वारा कनेक्शन को सही ढंग से अलग करता है | -| 🔄**इलेक्ट्रॉन ऑटो-अपडेट** | डेस्कटॉप ऐप अपडेट की जांच करता है + रीस्टार्ट होने पर ऑटो-इंस्टॉल | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| फ़ीचर | यह क्या करता है | -| --------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**एमसीपी सर्वर (25 उपकरण)** | 3 ट्रांसपोर्ट के माध्यम से आईडीई/एजेंट उपकरण: stdio, SSE (`/api/mcp/sse`), स्ट्रीम करने योग्य HTTP (`/api/mcp/stream`)। 18 कोर + 3 मेमोरी + 4 कौशल उपकरण | -| 🤝**A2A सर्वर (JSON-RPC + SSE)** | सिंक और स्ट्रीमिंग प्रवाह के साथ एजेंट-टू-एजेंट कार्य निष्पादन | -| 🧭**समेकित समापन बिंदु पृष्ठ** | एंडपॉइंट प्रॉक्सी, एमसीपी, ए2ए और एपीआई एंडपॉइंट टैब के साथ टैब्ड प्रबंधन पृष्ठ | -| 🎚️**सेवा सक्षम/अक्षम टॉगल** | सेटिंग्स दृढ़ता के साथ एमसीपी और ए2ए के लिए चालू/बंद स्विच (डिफ़ॉल्ट: बंद) | -| 🛰️**एमसीपी रनटाइम हार्टबीट** | वास्तविक प्रक्रिया स्थिति (पीआईडी, अपटाइम, दिल की धड़कन की उम्र, परिवहन, स्कोप मोड) | -| 📋**एमसीपी ऑडिट ट्रेल** | सफलता/असफलता और मुख्य एट्रिब्यूशन के साथ फ़िल्टर करने योग्य ऑडिट लॉग | -| 🔐**एमसीपी स्कोप प्रवर्तन** | नियंत्रित टूल एक्सेस के लिए 10 ग्रैन्युलर स्कोप अनुमतियाँ | -| 📡**A2A कार्य जीवनचक्र प्रबंधन** | कार्यों को सूचीबद्ध करें/फ़िल्टर करें, घटनाओं/कलाकृतियों का निरीक्षण करें, चल रहे कार्यों को रद्द करें | -| 📋**एजेंट कार्ड डिस्कवरी** | क्लाइंट ऑटो-डिस्कवरी के लिए `/.well-known/agent.json` | -| 🧪**प्रोटोकॉल E2E टेस्ट हार्नेस** | वास्तविक MCP SDK + A2A क्लाइंट `test:protocols:e2e` | में प्रवाहित होता है | -| ⚙️**परिचालन नियंत्रण** | कॉम्बो स्विच करें, लचीलापन प्रोफ़ाइल लागू करें, एक नियंत्रण सतह से ब्रेकर रीसेट करें | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| फ़ीचर | यह क्या करता है | -| -------------------------------- | ------------------------------------------------------------------------- | ----------------------- | -| 🎯**स्मार्ट 4-टियर फ़ॉलबैक** | Auto-route: Subscription → API Key → Cheap → Free | -| 📊**वास्तविक समय कोटा ट्रैकिंग** | लाइव टोकन गिनती + प्रति प्रदाता रीसेट उलटी गिनती | -| 🔄**प्रारूप अनुवाद** | OpenAI ↔ क्लाउड ↔ जेमिनी ↔ स्कीमा-सुरक्षित रूपांतरण के साथ प्रतिक्रियाएँ | -| 👥**मल्टी-अकाउंट सपोर्ट** | बुद्धिमान चयन के साथ प्रति प्रदाता एकाधिक खाते | -| 🔄**ऑटो टोकन रिफ्रेश** | OAuth टोकन पुनः प्रयास के साथ स्वचालित रूप से ताज़ा हो जाते हैं | -| 🎨**कस्टम कॉम्बो** | 9 संतुलन रणनीतियाँ + फ़ॉलबैक श्रृंखला नियंत्रण | -| 🌐**वाइल्डकार्ड राउटर** | `प्रदाता/*` डायनेमिक रूटिंग | -| 🧠**सोच बजट नियंत्रण** | पासथ्रू, ऑटो, कस्टम और अनुकूली तर्क सीमाएँ | -| 🔀**मॉडल उपनाम** | बिल्ट-इन + कस्टम मॉडल अलियासिंग और माइग्रेशन सुरक्षा | -| ⚡**पृष्ठभूमि का क्षरण** | कम प्राथमिकता वाले पृष्ठभूमि कार्यों को सस्ते मॉडल पर रूट करें | -| 🧪**टास्क-अवेयर स्मार्ट रूटिंग** | सामग्री प्रकार (कोडिंग/विज़न/विश्लेषण/सारांशीकरण) द्वारा स्वतः-चयन मॉडल | -| 🔄**A2A एजेंट वर्कफ़्लोज़** | स्टेटफुल मल्टी-स्टेप एजेंट निष्पादन के लिए नियतात्मक एफएसएम ऑर्केस्ट्रेटर | -| 🔀**अनुकूली रूटिंग** | टोकन वॉल्यूम और शीघ्र जटिलता के आधार पर गतिशील रणनीति ओवरराइड | -| 🎲**प्रदाता विविधता** | शैनन एन्ट्रापी स्कोरिंग संतुलन ऑटो-कॉम्बो ट्रैफ़िक वितरण | -| 💬**सिस्टम प्रॉम्प्ट इंजेक्शन** | वैश्विक व्यवहार नियंत्रण लगातार लागू | -| 📄**प्रतिक्रियाएं एपीआई संगतता** | कोडेक्स और उन्नत एजेंटिक वर्कफ़्लोज़ के लिए पूर्ण `/v1/responses` समर्थन | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| फ़ीचर | यह क्या करता है | -| --------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**छवि निर्माण** | `/v1/images/जेनरेशन` क्लाउड और स्थानीय बैकएंड के साथ | -| 📐**एंबेडिंग** | खोज और RAG पाइपलाइनों के लिए `/v1/embeddings` | -| 🎤**ऑडियो ट्रांस्क्रिप्शन** | `/v1/ऑडियो/ट्रांसक्रिप्शन` - 7 प्रदाता (डीपग्राम नोवा 3, असेंबलीएआई, ग्रोक व्हिस्पर, हगिंगफेस, इलेवनलैब्स, ओपनएआई, एज़्योर), ऑटो-लैंग्वेज डिटेक्शन, एमपी4/एमपी3/डब्ल्यूएवी सपोर्ट | -| 🔊**टेक्स्ट-टू-स्पीच** | `/v1/ऑडियो/स्पीच` - सही त्रुटि संदेशों के साथ 10 प्रदाता (इलेवनलैब्स, ओपनएआई, डीपग्राम, कार्टेसिया, प्लेएचटी, हगिंगफेस, एनवीडिया एनआईएम, इनवर्ल्ड, कोक्वी, टोर्टोइज़) | -| 🎬**वीडियो जेनरेशन** | `/v1/वीडियो/पीढ़ी` (ComfyUI + SD WebUI वर्कफ़्लोज़) | -| 🎵**संगीत पीढ़ी** | `/v1/संगीत/पीढ़ी` (ComfyUI वर्कफ़्लोज़) | -| 🛡️**संयम** | `/v1/मॉडरेशन` सुरक्षा जांच | -| 🔀**पुनर्रैंकिंग** | प्रासंगिकता स्कोरिंग के लिए `/v1/rerank` | -| 🔍**वेब खोज**🆕 | `/v1/search` - 5 प्रदाता (सर्पर, ब्रेव, पर्प्लेक्सिटी, एक्सा, टैविली), 6,500+ मुफ़्त/माह, ऑटो-फ़ेलओवर, कैश | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| फ़ीचर | यह क्या करता है | -| ------------------------------------------ | ----------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------- | -| 🔌**सर्किट तोड़ने वाले** | प्रति-मॉडल यात्रा/सीमा नियंत्रण के साथ पुनर्प्राप्ति | -| 🎯**एंडपॉइंट-अवेयर मॉडल** | कस्टम मॉडल समर्थित एंडपॉइंट + एपीआई प्रारूप की घोषणा करते हैं | -| 🛡️**एंटी-थंडरिंग झुंड** | पुनः प्रयास/दर घटनाओं पर म्यूटेक्स + सेमाफोर सुरक्षा | -| 🧠**सिमेंटिक + सिग्नेचर कैश** | दो कैश परतों के साथ लागत/विलंबता में कमी | -| ⚡**निष्क्रियता का अनुरोध** | डुप्लिकेट सुरक्षा विंडो | -| 🔒**टीएलएस फ़िंगरप्रिंट स्पूफिंग** | ब्राउज़र जैसा टीएलएस फ़िंगरप्रिंट -**बॉट डिटेक्शन और अकाउंट फ़्लैगिंग को कम करता है** | -| 🔏**सीएलआई फ़िंगरप्रिंट मिलान** | मूल सीएलआई अनुरोध हस्ताक्षरों से मेल खाता है -**प्रॉक्सी आईपी को संरक्षित करते हुए प्रतिबंध जोखिम को कम करता है** | -| 🌐**आईपी फ़िल्टरिंग** | उजागर तैनाती के लिए अनुमति सूची/अवरुद्ध सूची नियंत्रण | -| 📊**संपादन योग्य दर सीमाएँ** | दृढ़ता के साथ कॉन्फ़िगर करने योग्य वैश्विक/प्रदाता-स्तर की सीमाएं | -| 📉**सौम्य पतन** | मल्टी-लेयर क्षमता फ़ॉलबैक कोर गेटवे ऑपरेशंस की सुरक्षा करती है | -| 📜**कॉन्फिग ऑडिट ट्रेल** | डिफ-आधारित परिवर्तन ट्रैकिंग सरल रोलबैक के साथ परिचालन बहाव को रोकती है | -| ⏳**प्रदाता स्वास्थ्य सिंक** | प्राधिकरण विफलताओं से पहले ट्रिगरिंग अलर्ट सक्रिय टोकन समाप्ति निगरानी | -| 🚪**प्रतिबंधित खातों को स्वतः अक्षम करें** | ऑपरेशनल सर्किट ब्रेकर स्वचालित रूप से स्थायी रूप से ब्लॉक किए गए टोकन खातों को सील कर देता है | -| 🔑**एपीआई कुंजी प्रबंधन + स्कोपिंग** | सुरक्षित कुंजी जारी करना/रोटेशन और मॉडल/प्रदाता नियंत्रण | -| 👁️**स्कोप्ड एपीआई कुंजी का खुलासा**🆕 | `ALLOW_API_KEY_REVEAL` | के माध्यम से एपीआई कुंजियों की ऑप्ट-इन पुनर्प्राप्ति | -| 🛡️**संरक्षित `/मॉडल`** | मॉडल कैटलॉग के लिए वैकल्पिक प्रमाणीकरण गेटिंग और प्रदाता छिपाना | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| फ़ीचर | यह क्या करता है | -| ---------------------------------- | ----------------------------------------------------------------------- | ---------------------------- | -| 📝**अनुरोध + प्रॉक्सी लॉगिंग** | पूर्ण अनुरोध/प्रतिक्रिया और प्रॉक्सी लॉगिंग | -| 📉**स्ट्रीम किए गए विस्तृत लॉग**🆕 | एसएसई पेलोड स्ट्रीम को यूआई में साफ-सुथरा तरीके से पुनर्निर्माण करता है | -| 📋**एकीकृत लॉग डैशबोर्ड** | एक पृष्ठ में अनुरोध, प्रॉक्सी, ऑडिट और कंसोल दृश्य | -| 🔍**टेलीमेट्री के लिए अनुरोध** | p50/p95/p99 विलंबता और अनुरोध अनुरेखण | -| 🏥**स्वास्थ्य डैशबोर्ड** | अपटाइम, ब्रेकर स्थिति, लॉकआउट, कैश आँकड़े | -| 💰**लागत ट्रैकिंग** | बजट नियंत्रण और प्रति-मॉडल मूल्य निर्धारण दृश्यता | -| 📈**एनालिटिक्स विज़ुअलाइज़ेशन** | मॉडल/प्रदाता उपयोग अंतर्दृष्टि और रुझान दृश्य | -| 🧪**मूल्यांकन ढाँचा** | विन्यास योग्य मिलान रणनीतियों के साथ गोल्डन सेट परीक्षण | -| 📡**लाइव डायग्नोस्टिक्स**🆕 | सटीक कॉम्बो लाइव परीक्षण के लिए सिमेंटिक कैश बाईपास | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| फ़ीचर | यह क्या करता है | -| --------------------------------- | ----------------------------------------------------------------------------------- | ------------------------------------ | -| 🌐**कहीं भी तैनात करें** | लोकलहोस्ट, वीपीएस, डॉकर, क्लाउड वातावरण | -| 🚇**क्लाउडफ्लेयर टनल**🆕 | डैशबोर्ड से एक-क्लिक त्वरित सुरंग एकीकरण | -| 🔑**एपीआई कुंजी मॉडल फ़िल्टरिंग** | मूल /v1/मॉडल प्रतिक्रिया निर्दिष्ट बियरर संदर्भ भूमिकाओं के माध्यम से फ़िल्टर की गई | -| ⚡**स्मार्ट कैश बायपास** | कॉन्फ़िगर करने योग्य टीटीएल अनुमान और फ़ोर्स्ड रीफ़ेच नियंत्रण | -| 🔄**बैकअप/पुनर्स्थापना** | निर्यात/आयात और आपदा पुनर्प्राप्ति प्रवाह | -| 🧙**ऑनबोर्डिंग विज़ार्ड** | फर्स्ट-रन गाइडेड सेटअप | -| 🔧**सीएलआई टूल्स डैशबोर्ड** | लोकप्रिय कोडिंग टूल के लिए एक-क्लिक सेटअप | -| 🎮**मॉडल खेल का मैदान** | डैशबोर्ड से किसी भी प्रदाता/मॉडल/एंडपॉइंट का परीक्षण करें | -| 🔏**सीएलआई फ़िंगरप्रिंट टॉगल** | सेटिंग्स > सुरक्षा | में प्रति प्रदाता फ़िंगरप्रिंट मिलान | -| 🌐**i18n (30 भाषाएँ)** | पूर्ण डैशबोर्ड + आरटीएल कवरेज के साथ डॉक्स भाषा समर्थन | -| 🧹**सभी मॉडल साफ़ करें** | प्रदाता विवरण में एक-क्लिक मॉडल सूची समाशोधन | -| 👁️**साइडबार नियंत्रण**🆕 | उपस्थिति सेटिंग्स से घटकों और एकीकरणों को छुपाएं | -| 📋**मुद्दा टेम्पलेट** | बग और सुविधाओं के लिए मानकीकृत GitHub टेम्पलेट | -| 📂**कस्टम डेटा निर्देशिका** | भंडारण स्थान के लिए `DATA_DIR` ओवरराइड | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1292,103 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -जब कोटा, दर, या स्वास्थ्य विफल हो जाता है, तो ओमनीरूट मैन्युअल स्विचिंग के बिना स्वचालित रूप से अगले उम्मीदवार के पास चला जाता है।#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A UI और डॉक्स में खोजने योग्य हैं (छिपा हुआ नहीं) -- प्रोटोकॉल स्थिति एपीआई लाइव परिचालन डेटा को उजागर करते हैं (`/api/mcp/*`, `/api/a2a/*`) -- डैशबोर्ड में दिन-2 ऑप्स के लिए क्रियाएं शामिल हैं (कॉम्बो टॉगल, ब्रेकर रीसेट, कार्य रद्द करना)#### Translator + validation workflow +#### Protocol management that is visible and operable -अनुवादक क्षेत्र में शामिल हैं: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**खेल का मैदान**: परिवर्तन जांच का अनुरोध करें -**चैट परीक्षक**: पूर्ण अनुरोध/प्रतिक्रिया राउंड-ट्रिप -**टेस्ट बेंच**: एक बार में कई मामले -**लाइव मॉनिटर**: वास्तविक समय यातायात दृश्य +#### Translator + validation workflow -साथ ही `npm run test:protocols:e2e` के माध्यम से वास्तविक ग्राहकों के साथ प्रोटोकॉल सत्यापन। +The Translator area includes: -> 📖**[एमसीपी सर्वर रीडमी](ओपन-एसएसई/एमसीपी-सर्वर/रीडमी.एमडी)**- टूल संदर्भ, आईडीई कॉन्फ़िगरेशन और क्लाइंट उदाहरण +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A सर्वर README](src/lib/a2a/README.md)**- कौशल, JSON-RPC विधियाँ, स्ट्रीमिंग, और कार्य जीवनचक्र## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -ओमनीरूट में गोल्डन सेट के मुकाबले एलएलएम प्रतिक्रिया गुणवत्ता का परीक्षण करने के लिए एक अंतर्निहित मूल्यांकन ढांचा शामिल है। डैशबोर्ड में**एनालिटिक्स → इवेल्स**के माध्यम से इसे एक्सेस करें।### Built-in Golden Set +## 🧪 Evaluations (Evals) -प्री-लोडेड "ओम्नीरूट गोल्डन सेट" में इसके लिए परीक्षण मामले शामिल हैं: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- नमस्ते, गणित, भूगोल, कोड जनरेशन -- JSON प्रारूप अनुपालन, अनुवाद, मार्कडाउन पीढ़ी -- सुरक्षा इनकार (हानिकारक सामग्री), गिनती, बूलियन तर्क### Evaluation Strategies +### Built-in Golden Set -| रणनीति | विवरण | उदाहरण | -| ---------- | ------------------------------------------------- | ------------------------------ | --- | -| 'सटीक' | आउटपुट बिल्कुल मेल खाना चाहिए | `"4"` | -| 'शामिल है' | आउटपुट में सबस्ट्रिंग (केस-असंवेदनशील) होना चाहिए | `"पेरिस"` | -| 'रेगेक्स' | आउटपुट रेगेक्स पैटर्न से मेल खाना चाहिए | `"1.*2.*3"` | -| `कस्टम` | कस्टम जेएस फ़ंक्शन सही/गलत लौटाता है | `(आउटपुट) => आउटपुट.लेंथ > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) -<विवरण> -<सारांश>🧩 एमसीपी सेटअप (मॉडल संदर्भ प्रोटोकॉल) +
+🧩 MCP Setup (Model Context Protocol) -stdio मोड में MCP ट्रांसपोर्ट प्रारंभ करें:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -अनुशंसित सत्यापन प्रवाह: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. अपने MCP क्लाइंट को stdio से कनेक्ट करें। -2. `omniroute_get_health` चलाएँ। -3. `omniroute_list_combos` चलाएँ। -4. दिल की धड़कन, गतिविधि और ऑडिट की पुष्टि करने के लिए `/डैशबोर्ड/एमसीपी` खोलें। +Useful APIs for automation: -स्वचालन के लिए उपयोगी एपीआई: +- `GET /api/mcp/status` +- `GET /api/mcp/tools` +- `GET /api/mcp/audit` +- `GET /api/mcp/audit/stats` -- `प्राप्त करें /एपीआई/एमसीपी/स्थिति` -- `प्राप्त करें /एपीआई/एमसीपी/टूल्स` -- `प्राप्त करें /एपीआई/एमसीपी/ऑडिट` -- `प्राप्त करें /api/mcp/ऑडिट/आँकड़े`
+ -<विवरण> -<सारांश>🤝 A2A सेटअप (एजेंट2एजेंट) +
+🤝 A2A Setup (Agent2Agent) -एजेंट का पता लगाएं:```bash +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -एक कार्य भेजें:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` +Manage lifecycle: -जीवनचक्र प्रबंधित करें: - -- `प्राप्त करें /api/a2a/status` -- `प्राप्त करें /api/a2a/कार्य` -- `प्राप्त करें /api/a2a/tasks/:id` +- `GET /api/a2a/status` +- `GET /api/a2a/tasks` +- `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -परिचालन यूआई: +Operational UI: -- कार्य/स्थिति/स्ट्रीम अवलोकन और धूम्रपान क्रियाओं के लिए `/डैशबोर्ड/ए2ए`
+- `/dashboard/a2a` for task/state/stream observability and smoke actions -<विवरण> -<सारांश>🧪 एंड-टू-एंड प्रोटोकॉल सत्यापन + -वास्तविक ग्राहकों के साथ दोनों प्रोटोकॉल मान्य करें:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -यह सत्यापित करता है: +This verifies: -- एमसीपी एसडीके क्लाइंट कनेक्ट/लिस्ट/कॉल -- A2A खोज/भेजें/स्ट्रीम/प्राप्त करें/रद्द करें -- एमसीपी ऑडिट और ए2ए कार्य प्रबंधन एपीआई में डेटा को क्रॉस-चेक करें
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs -<विवरण> -<सारांश>💳 सदस्यता प्रदाता### Claude Code (Pro/Max) + + +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1401,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**प्रो टिप:**जटिल कार्यों के लिए ओपस और गति के लिए सॉनेट का उपयोग करें। ओमनीरूट प्रति मॉडल कोटा ट्रैक करता है!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1415,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -प्रत्येक कोडेक्स खाते में अब `डैशबोर्ड -> प्रदाता` में नीति टॉगल हैं: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5 घंटे` (चालू/बंद): 5 घंटे की विंडो सीमा नीति लागू करें। -- `साप्ताहिक` (चालू/बंद): साप्ताहिक विंडो सीमा नीति लागू करें। -- थ्रेसहोल्ड व्यवहार: जब एक सक्षम विंडो >=90% उपयोग तक पहुंच जाती है, तो वह खाता छोड़ दिया जाता है। -- रोटेशन व्यवहार: ओमनीरूट स्वचालित रूप से अगले पात्र कोडेक्स खाते पर रूट करता है। -- रीसेट व्यवहार: जब प्रदाता का `resetAt` समय बीत जाता है, तो खाता स्वचालित रूप से फिर से पात्र हो जाता है। +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -परिदृश्य: +Scenarios: -- `5 घंटे चालू` + `साप्ताहिक चालू`: जब कोई भी विंडो सीमा तक पहुंचती है तो खाता छोड़ दिया जाता है। -- `5 घंटे की छूट` + `साप्ताहिक चालू`: केवल साप्ताहिक उपयोग ही खाते को ब्लॉक कर सकता है। -- `5 घंटे चालू` + `साप्ताहिक बंद`: केवल 5 घंटे का उपयोग ही खाते को ब्लॉक कर सकता है। -- `resetAt` पारित: खाता स्वचालित रूप से रोटेशन में पुनः प्रवेश करता है (कोई मैन्युअल पुनः सक्षम नहीं)।### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1440,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**सर्वोत्तम मूल्य:**विशाल निःशुल्क स्तर! सशुल्क स्तरों से पहले इसका उपयोग करें।### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1455,71 +1662,91 @@ Models:
-<विवरण> -<सारांश>🔑 एपीआई कुंजी प्रदाता### NVIDIA NIM (FREE developer access — 70+ models) +
+🔑 API Key Providers -1. साइन अप करें: [build.nvidia.com](https://build.nvidia.com) -2. निःशुल्क एपीआई कुंजी प्राप्त करें (1000 अनुमान क्रेडिट शामिल) -3. डैशबोर्ड → प्रदाता जोड़ें → एनवीडिया एनआईएम: - - एपीआई कुंजी: `nvapi-your-key` +### NVIDIA NIM (FREE developer access — 70+ models) -**मॉडल:**`एनवीडिया/लामा-3.3-70बी-इंस्ट्रक्ट`, `एनवीडिया/मिस्ट्रल-7बी-इंस्ट्रक्ट`, और 50+ अधिक +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**प्रो टिप:**ओपनएआई-संगत एपीआई - ओमनीरूट के प्रारूप अनुवाद के साथ सहजता से काम करता है!### DeepSeek +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -1. साइन अप करें: [प्लेटफ़ॉर्म.डीपसीक.कॉम](https://platform.डीपसीक.कॉम) -2. एपीआई कुंजी प्राप्त करें -3. डैशबोर्ड → प्रदाता जोड़ें → डीपसीक +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -**मॉडल:**`डीपसीक/डीपसीक-चैट`, `डीपसीक/डीपसीक-कोडर`### Groq (Free Tier Available!) +### DeepSeek -1. साइन अप करें: [console.groq.com](https://console.groq.com) -2. एपीआई कुंजी प्राप्त करें (फ्री टियर शामिल) -3. डैशबोर्ड → प्रदाता जोड़ें → ग्रोक +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -**मॉडल:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**प्रो टिप:**अल्ट्रा-फास्ट अनुमान - वास्तविक समय कोडिंग के लिए सर्वोत्तम!### OpenRouter (100+ Models) +### Groq (Free Tier Available!) -1. साइन अप करें: [openrouter.ai](https://openrouter.ai) -2. एपीआई कुंजी प्राप्त करें -3. डैशबोर्ड → प्रदाता जोड़ें → ओपनराउटर +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -**मॉडल:**एक ही एपीआई कुंजी के माध्यम से सभी प्रमुख प्रदाताओं से 100+ मॉडल तक पहुंचें। +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**डैशबोर्ड व्यवहार:**ओपनराउटर मॉडल**उपलब्ध मॉडल**से प्रबंधित किए जाते हैं। मैन्युअल ऐड, आयात और ऑटो-सिंक सभी एक ही सूची को अपडेट करते हैं।
+**Pro Tip:** Ultra-fast inference — best for real-time coding! -<विवरण> -<सारांश>💰 सस्ते प्रदाता (बैकअप)### GLM-4.7 (Daily reset, $0.6/1M) +### OpenRouter (100+ Models) -1. साइन अप करें: [झिपु एआई](https://open.bigmodel.cn/) -2. कोडिंग योजना से एपीआई कुंजी प्राप्त करें -3. डैशबोर्ड → एपीआई कुंजी जोड़ें: - - प्रदाता: `glm` - - एपीआई कुंजी: `आपकी-कुंजी` +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -**उपयोग:**`glm/glm-4.7` +**Models:** Access 100+ models from all major providers through a single API key. -**प्रो टिप:**कोडिंग प्लान 1/7 लागत पर 3× कोटा प्रदान करता है! प्रतिदिन सुबह 10:00 बजे रीसेट करें।### MiniMax M2.1 (5h reset, $0.20/1M) +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -1. साइन अप करें: [मिनीमैक्स](https://www.minimax.io/) -2. एपीआई कुंजी प्राप्त करें -3. डैशबोर्ड → एपीआई कुंजी जोड़ें + -**उपयोग करें:**`मिनीमैक्स/मिनीमैक्स-एम2.1` +
+💰 Cheap Providers (Backup) -**Pro Tip:**Cheapest option for long context (1M tokens)!### Kimi K2 ($9/month flat) +### GLM-4.7 (Daily reset, $0.6/1M) -1. सदस्यता लें: [मूनशॉट एआई](https://platform.moonshot.ai/) -2. एपीआई कुंजी प्राप्त करें -3. डैशबोर्ड → एपीआई कुंजी जोड़ें +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**उपयोग करें:**`किमी/किमी-नवीनतम` +**Use:** `glm/glm-4.7` -**प्रो टिप:**10एम टोकन के लिए निश्चित $9/माह = $0.90/1एम प्रभावी लागत!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. -<विवरण> -<सारांश>🆓 मुफ़्त प्रदाता (आपातकालीन बैकअप)### Qoder (5 FREE models via OAuth) +### MiniMax M2.1 (5h reset, $0.20/1M) + +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `minimax/MiniMax-M2.1` + +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1560,8 +1787,10 @@ Models:
-<विवरण> -<सारांश>🎨 कॉम्बो बनाएं### Example 1: Maximize Subscription → Cheap Backup +
+🎨 Create Combos + +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1589,8 +1818,10 @@ Cost: $0 forever!
-<विवरण> -<सारांश>🔧 सीएलआई एकीकरण### Cursor IDE +
+🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1601,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -एक-क्लिक कॉन्फ़िगरेशन के लिए डैशबोर्ड में**सीएलआई टूल्स**पृष्ठ का उपयोग करें, या `~/.claude/settings.json` को मैन्युअल रूप से संपादित करें।### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1612,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**विकल्प 1 - डैशबोर्ड (अनुशंसित):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**विकल्प 2 - मैनुअल:**`~/.openclaw/openclaw.json` संपादित करें:```json +```json { "models": { "providers": { @@ -1629,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **ध्यान दें:**ओपनक्लाव केवल स्थानीय ओमनीरूट के साथ काम करता है। IPv6 रिज़ॉल्यूशन समस्याओं से बचने के लिए `लोकलहोस्ट` के बजाय `127.0.0.1` का उपयोग करें।### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1643,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**चरण 1:**एक कस्टम प्रदाता के रूप में ओम्निरूट जोड़ें:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**चरण 2:**अपने प्रोजेक्ट रूट में `opencode.json` बनाएं/संपादित करें:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1669,118 +1909,130 @@ opencode } } } -```` +``` -**चरण 3:**ओपनकोड में मॉडल का चयन करें:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**टिप:**अपने ओमनीरूट `/v1/models` एंडपॉइंट में उपलब्ध किसी भी मॉडल को `मॉडल` अनुभाग में जोड़ें। अपने ओमनीरूट डैशबोर्ड से `प्रदाता/मॉडल-आईडी` प्रारूप का उपयोग करें।
+ --- ## समस्या निवारण -<विवरण> -<सारांश>समस्या निवारण मार्गदर्शिका का विस्तार करने के लिए क्लिक करें +
+Click to expand troubleshooting guide -**"भाषा मॉडल ने संदेश प्रदान नहीं किया"** +**"Language model did not provide messages"** -- प्रदाता कोटा समाप्त → डैशबोर्ड कोटा ट्रैकर की जाँच करें -- समाधान: कॉम्बो फ़ॉलबैक का उपयोग करें या सस्ते स्तर पर स्विच करें +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**दर सीमित करना** +**Rate limiting** -- सदस्यता कोटा ख़त्म → GLM/MiniMax पर फ़ॉलबैक -- कॉम्बो जोड़ें: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**OAuth टोकन समाप्त हो गया** +**OAuth token expired** -- ओम्निरूट द्वारा स्वतः ताज़ा -- यदि समस्या बनी रहती है: डैशबोर्ड → प्रदाता → पुनः कनेक्ट करें +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**उच्च लागत** +**High costs** -- डैशबोर्ड → लागत में उपयोग के आँकड़े जाँचें -- प्राथमिक मॉडल को जीएलएम/मिनीमैक्स पर स्विच करें -- गैर-महत्वपूर्ण कार्यों के लिए फ्री टियर (मिथुन सीएलआई, क्यूडर) का उपयोग करें +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**डैशबोर्ड/एपीआई पोर्ट गलत हैं** +**Dashboard/API ports are wrong** -- `पोर्ट` कैनोनिकल बेस पोर्ट है (और डिफ़ॉल्ट रूप से एपीआई पोर्ट) -- `API_PORT` केवल OpenAI-संगत API श्रोता को ओवरराइड करता है -- `DASHBOARD_PORT` केवल डैशबोर्ड/Next.js श्रोता को ओवरराइड करता है -- अपने डैशबोर्ड/सार्वजनिक URL पर `NEXT_PUBLIC_BASE_URL` सेट करें (OAuth कॉलबैक के लिए) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**क्लाउड सिंक त्रुटियाँ** +**Cloud sync errors** -- अपने चल रहे उदाहरण के लिए `BASE_URL` बिंदुओं को सत्यापित करें -- अपने अपेक्षित क्लाउड एंडपॉइंट पर `CLOUD_URL` बिंदुओं को सत्यापित करें -- `NEXT_PUBLIC_*` मानों को सर्वर-साइड मानों के साथ संरेखित रखें +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**पहला लॉगिन काम नहीं कर रहा** +**First login not working** -- `.env` में `INITIAL_PASSWORD` जांचें -- यदि सेट नहीं है, तो फ़ॉलबैक पासवर्ड `123456` है +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**कोई अनुरोध लॉग नहीं** +**No request logs** -- अनुरोध कलाकृतियों को प्रति अनुरोध एक JSON फ़ाइल के रूप में `DATA_DIR/call_logs/` पर लिखा जाता है -- यदि आपको विस्तृत प्रति-स्टेज पेलोड की आवश्यकता है तो डैशबोर्ड → लॉग → अनुरोध लॉग से पाइपलाइन कैप्चर सक्षम करें -- यदि आप `logs/application/app.log` में एप्लिकेशन कंसोल लॉग भी चाहते हैं तो `APP_LOG_TO_FILE=true` सेट करें -- `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, और `CALL_LOG_MAX_ENTRIES` को आवश्यकतानुसार समायोजित करें +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**कनेक्शन परीक्षण OpenAI-संगत प्रदाताओं के लिए "अमान्य" दिखाता है** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- कई प्रदाता `/मॉडल` समापन बिंदु को उजागर नहीं करते हैं -- ओमनीरूट v1.0.6+ में चैट पूर्णता के माध्यम से फ़ॉलबैक सत्यापन शामिल है -- सुनिश्चित करें कि आधार URL में `/v1` प्रत्यय शामिल है### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix - - +### 🔐 OAuth on a Remote Server ->**⚠️ वीपीएस, डॉकर या किसी रिमोट सर्वर पर ओमनीरूट चलाने वाले उपयोगकर्ताओं के लिए महत्वपूर्ण**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? + + -**एंटीग्रेविटी**और**जेमिनी सीएलआई**प्रदाता**Google OAuth 2.0**का उपयोग करते हैं। Google को ऐप के Google क्लाउड कंसोल में पूर्व-पंजीकृत यूआरआई में से एक से सटीक मिलान करने के लिए OAuth प्रवाह में `redirect_uri` की आवश्यकता है। +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -ओम्निरूट में बंडल किए गए OAuth क्रेडेंशियल**केवल `लोकलहोस्ट`* के लिए पंजीकृत हैं। जब आप किसी दूरस्थ सर्वर (उदाहरण के लिए `https://omniroute.myserver.com`) पर ओमनीरूट एक्सेस करते हैं, तो Google प्रमाणीकरण को अस्वीकार कर देता है:``` +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? + +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -आपको अपने सर्वर के यूआरआई के साथ Google क्लाउड कंसोल में एक**OAuth 2.0 क्लाइंट आईडी**बनाना होगा।#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Google क्लाउड कंसोल खोलें** +#### Step-by-step -यहां जाएं: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. एक नया OAuth 2.0 क्लाइंट आईडी बनाएं** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) --**"+ क्रेडेंशियल बनाएं"**→**"OAuth क्लाइंट आईडी"**पर क्लिक करें +**2. Create a new OAuth 2.0 Client ID** -- एप्लिकेशन प्रकार:**"वेब एप्लिकेशन"** -- नाम: कुछ भी जो आपको पसंद हो (जैसे `ओम्नीरूट रिमोट`) +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -**3. अधिकृत रीडायरेक्ट यूआरआई जोड़ें** +**3. Add Authorized Redirect URIs** -**"अधिकृत रीडायरेक्ट यूआरआई"**फ़ील्ड में, जोड़ें:``` +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> `your-server.com` को अपने सर्वर के डोमेन या आईपी से बदलें (यदि आवश्यक हो तो पोर्ट शामिल करें, उदाहरण के लिए `http://45.33.32.156:20128/callback`)। +**4. Save and copy the credentials** -**4. क्रेडेंशियल सहेजें और कॉपी करें** +After creating, Google will show the **Client ID** and **Client Secret**. -बनाने के बाद, Google**क्लाइंट आईडी**और**क्लाइंट सीक्रेट**दिखाएगा। +**5. Set environment variables** -**5. पर्यावरण चर सेट करें** +In your `.env` (or Docker environment variables): -आपके `.env` (या डॉकर पर्यावरण चर) में:```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1789,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. ओम्निरूट को पुनरारंभ करें**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. पुनः कनेक्ट करने का प्रयास करें** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -डैशबोर्ड → प्रदाता → एंटीग्रेविटी (या जेमिनी सीएलआई) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google अब `https://your-server.com/callback` पर सही ढंग से रीडायरेक्ट करेगा।--- +--- #### Temporary workaround (without custom credentials) -यदि आप अभी अपना स्वयं का क्रेडेंशियल सेट नहीं करना चाहते हैं, तो आप अभी भी**मैन्युअल यूआरएल प्रवाह**का उपयोग कर सकते हैं: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. ओमनीरूट Google प्राधिकरण URL खोलता है -2. अधिकृत करने के बाद, Google `लोकलहोस्ट` पर रीडायरेक्ट करने का प्रयास करता है (जो रिमोट सर्वर पर विफल रहता है) -3.**अपने ब्राउज़र के एड्रेस बार से पूरा यूआरएल कॉपी करें**(भले ही पेज लोड न हो) -4. उस यूआरएल को ओमनीरूट कनेक्शन मोडल में दिखाए गए फ़ील्ड में पेस्ट करें -5.**"कनेक्ट"**पर क्लिक करें +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> यह काम करता है क्योंकि यूआरएल में प्राधिकरण कोड इस बात पर ध्यान दिए बिना मान्य है कि रीडायरेक्ट पेज लोड किया गया है या नहीं।--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. -<विवरण> -<सारांश>🇧🇷 पुर्तगाली भाषा में वर्साओ#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -प्रमाणित करने के लिए**एंटीग्रेविटी**और**मिथुन सीएलआई**का उपयोग करें**Google OAuth 2.0**का उपयोग करें। Google को एक `redirect_uri` का उपयोग करना चाहिए जो बिना किसी प्रवाह के OAuth सेजा**exatamente**का उपयोग करता है और आपको Google क्लाउड कंसोल के लिए यूआरआई डाउनलोड करने की आवश्यकता है। +
+🇧🇷 Versão em Português -जैसा कि OAuth का क्रेडेंशियल है, कोई ओम्निरूट एस्टाओ कैडस्ट्रास नहीं है**'लोकलहोस्ट'**के लिए एपेनास। एक सर्विडोर रिमोट (उदा: `https://omniroute.meuservidor.com`) पर ओम्निरूट का उपयोग कैसे करें, या Google एक ऑटेंटिका कॉम को पुनः प्राप्त करता है:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -आपका सटीक विवरण**OAuth 2.0 क्लाइंट आईडी**आपके सर्वर पर यूआरआई के साथ Google क्लाउड कंसोल नहीं है।#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Google क्लाउड कंसोल तक पहुंच** +#### Passo a passo -अब्राहम: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Acesse o Google Cloud Console** -**2. नया OAuth 2.0 क्लाइंट आईडी देखें** +Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- उन्हें क्लिक करें**"+ क्रेडेंशियल बनाएं"**→**"OAuth क्लाइंट आईडी"** -- आवेदन टिप:**"वेब एप्लिकेशन"** -- नोम: एस्कोल्हा क्वाल्कर नोम (उदा: `ओम्नीरूट रिमोट`) +**2. Crie um novo OAuth 2.0 Client ID** -**3. अधिकृत रीडायरेक्ट यूआरआई के रूप में एडिकियोन** +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -कोई शिकायत नहीं**"अधिकृत रीडायरेक्ट यूआरआई"**, आदि:``` +**3. Adicione as Authorized Redirect URIs** + +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> स्थानापन्न `seu-servidor.com` अपने आईपी को अपने सर्वर पर रखें (इसमें एक आवश्यक पोर्ट भी शामिल है, उदाहरण के लिए: `http://45.33.32.156:20128/callback`)। +**4. Salve e copie as credenciais** -**4. साख के रूप में सहेजें और कॉपी करें** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -एपोस क्रियर, Google द्वारा**क्लाइंट आईडी**और**क्लाइंट सीक्रेट**। +**5. Configure as variáveis de ambiente** -**5. परिवेश परिवर्तन**के रूप में कॉन्फ़िगर करें +No seu `.env` (ou nas variáveis de ambiente do Docker): -कोई सेउ `.env` (आप डॉकर के परिवेश को कैसे बदलते हैं):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1868,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. ओम्निरूट का नवीनीकरण**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute +``` -```` +**7. Tente conectar novamente** -**7. नए सिरे से संपर्क करें** +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -डैशबोर्ड → प्रदाता → एंटीग्रेविटी (या जेमिनी सीएलआई) → OAuth +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. -यदि आप `https://seu-servidor.com/callback` और एक प्रामाणिक कार्य के लिए Google पुनर्निर्देशन करते हैं।--- +--- #### Workaround temporário (sem configurar credenciais próprias) -यदि आप पहले से ही उचित क्रेडेंशियल प्राप्त नहीं करना चाहते हैं, तो आपके लिए फ्लक्सो का उपयोग करना संभव है**यूआरएल का मैनुअल**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. Google पर स्वचालित URL का उपयोग करके ओम्निरूट का उपयोग करें -2. आप स्वचालित रूप से काम कर सकते हैं, या Google `लोकलहोस्ट` को पुनः प्राप्त कर सकता है (यदि कोई सर्वर रिमोट नहीं है) -3.**एक यूआरएल को पूरा कॉपी करें**अपने ब्राउजर से दोबारा डाउनलोड करें (मुझे लगता है कि एक पेज अभी भी उपलब्ध है) -4. ओम्निरूट से जुड़ने के लिए कोई भी यूआरएल नहीं है -5. उन्हें क्लिक करें**"कनेक्ट"** +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) +4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute +5. Clique em **"Connect"** -> यह समाधान यूआरएल को स्वचालित रूप से डाउनलोड करने के लिए काम कर रहा है और आपके द्वारा किए गए रीडायरेक्ट को स्वतंत्र रूप से वैध बनाता है।
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1906,64 +2171,73 @@ docker restart omniroute ## 🛠️ Tech Stack -<विवरण> -<सारांश>तकनीकी स्टैक विवरण का विस्तार करने के लिए क्लिक करें +
+Click to expand tech stack details --**रनटाइम**: Node.js 18-22 LTS (⚠️ Node.js 24+**समर्थित नहीं**है - `better-sqlite3` मूल बायनेरिज़ असंगत हैं) --**भाषा**: टाइपस्क्रिप्ट 5.9 -**100% टाइपस्क्रिप्ट**`src/` और `open-sse/` में (v2.0 के बाद से कोर मॉड्यूल में शून्य `कोई भी`) --**फ्रेमवर्क**: नेक्स्ट.जेएस 16 + रिएक्ट 19 + टेलविंड सीएसएस 4 --**डेटाबेस**: LowDB (JSON) + SQLite (डोमेन स्थिति + प्रॉक्सी लॉग + MCP ऑडिट + रूटिंग निर्णय) --**स्कीमा**: ज़ॉड (एमसीपी टूल I/O सत्यापन, एपीआई अनुबंध) --**प्रोटोकॉल**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**स्ट्रीमिंग**: सर्वर-भेजे गए इवेंट (एसएसई) --**प्रामाणिक**: OAuth 2.0 (PKCE) + JWT + API कुंजियाँ + MCP स्कोप्ड प्राधिकरण --**परीक्षण**: Node.js टेस्ट रनर + विटेस्ट (यूनिट, एकीकरण, E2E सहित 900+ परीक्षण) --**सीआई/सीडी**: गिटहब क्रियाएँ (ऑटो एनपीएम प्रकाशन + रिलीज पर डॉकर हब) --**वेबसाइट**: [omniroute.online](https://omniroute.online) --**पैकेज**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**डॉकर**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**लचीलापन**: सर्किट ब्रेकर, एक्सपोनेंशियल बैकऑफ़, एंटी-थंडरिंग झुंड, टीएलएस स्पूफिंग, ऑटो-कॉम्बो सेल्फ-हीलिंग
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## दस्तावेज़ -| दस्तावेज़ | विवरण | -| ------------------------------------------------ | ---------------------------------------------------------------- | -| [उपयोगकर्ता गाइड](docs/USER_GUIDE.md) | प्रदाता, कॉम्बो, सीएलआई एकीकरण, तैनाती | -| [एपीआई संदर्भ](docs/API_REFERENCE.md) | उदाहरण सहित सभी समापन बिंदु | -| [एमसीपी सर्वर](ओपन-एसएसई/एमसीपी-सर्वर/रीडमी.एमडी) | 16 एमसीपी उपकरण, आईडीई कॉन्फ़िगरेशन, पायथन/टीएस/गो क्लाइंट | -| [ए2ए सर्वर](src/lib/a2a/README.md) | JSON-RPC 2.0 प्रोटोकॉल, कौशल, स्ट्रीमिंग, कार्य प्रबंधन | -| [ऑटो-कॉम्बो इंजन](docs/auto-combo.md) | 6-कारक स्कोरिंग, मोड पैक, स्व-उपचार | -| [समस्या निवारण](docs/TROUBLESHOOTING.md) | सामान्य समस्याएँ एवं समाधान | -| [आर्किटेक्चर](docs/ARCHITECTURE.md) | सिस्टम आर्किटेक्चर और आंतरिक | -| [योगदान](CONTRIBUTING.md) | विकास सेटअप और दिशानिर्देश | -| [OpenAPI Spec](docs/openapi.yaml) | ओपनएपीआई 3.0 विशिष्टता | -| [सुरक्षा नीति](सुरक्षा.एमडी) | भेद्यता रिपोर्टिंग और सुरक्षा प्रथाएं | -| [VM परिनियोजन](docs/VM_DEPLOYMENT_GUIDE.md) | संपूर्ण गाइड: VM + nginx + Cloudflare सेटअप | -| [फीचर्स गैलरी](docs/FEATURES.md) | स्क्रीनशॉट के साथ विजुअल डैशबोर्ड टूर | -| [रिलीज़ चेकलिस्ट](docs/RELEASE_CHECKLIST.md) | प्री-रिलीज़ सत्यापन चरण |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -ओम्निरूट ने कई विकास चरणों में**210+ सुविधाओं की योजना बनाई है**। यहां प्रमुख क्षेत्र हैं: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| श्रेणी | नियोजित विशेषताएं | हाइलाइट्स | -| -------------------------------- | ---------------- | -------------------------------------------------------------------------------------------------- | -| 🧠**रूटिंग और इंटेलिजेंस**| 25+ | न्यूनतम-विलंबता रूटिंग, टैग-आधारित रूटिंग, कोटा प्रीफ़्लाइट, पी2सी खाता चयन | -| 🔒**सुरक्षा एवं अनुपालन**| 20+ | एसएसआरएफ हार्डनिंग, क्रेडेंशियल क्लोकिंग, प्रति समापन बिंदु दर-सीमा, प्रबंधन कुंजी स्कोपिंग | -| 📊**अवलोकनशीलता**| 15+ | ओपन टेलीमेट्री एकीकरण, वास्तविक समय कोटा निगरानी, ​​प्रति मॉडल लागत ट्रैकिंग | -| 🔄**प्रदाता एकीकरण**| 20+ | डायनेमिक मॉडल रजिस्ट्री, प्रदाता कूलडाउन, मल्टी-अकाउंट कोडेक्स, कोपायलट कोटा पार्सिंग | -| ⚡**प्रदर्शन**| 15+ | दोहरी कैश परत, शीघ्र कैश, प्रतिक्रिया कैश, स्ट्रीमिंग कीपलाइव, बैच एपीआई | -| 🌐**पारिस्थितिकी तंत्र**| 10+ | वेबसॉकेट एपीआई, कॉन्फिग हॉट-रीलोड, वितरित कॉन्फिग स्टोर, वाणिज्यिक मोड |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**ओपनकोड इंटीग्रेशन**- ओपनकोड एआई कोडिंग आईडीई के लिए मूल प्रदाता समर्थन -- 🔗**TRAE एकीकरण**- TRAE AI विकास ढांचे के लिए पूर्ण समर्थन -- 📦**बैच एपीआई**- थोक अनुरोधों के लिए अतुल्यकालिक बैच प्रोसेसिंग -- 🎯**टैग-आधारित रूटिंग**- कस्टम टैग और मेटाडेटा के आधार पर रूट अनुरोध -- 💰**न्यूनतम-लागत रणनीति**— स्वचालित रूप से सबसे सस्ते उपलब्ध प्रदाता का चयन करें +### 🔜 Coming Soon -> 📝 पूर्ण सुविधा विशिष्टताएँ [`docs/new-features/`](docs/new-features/) में उपलब्ध हैं (217 विस्तृत विवरण)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1971,18 +2245,20 @@ docker restart omniroute ### How to Contribute -1. रिपॉजिटरी को फोर्क करें -2. अपनी फीचर शाखा बनाएं (`git checkout -b फीचर/अमेजिंग-फीचर`) -3. अपने परिवर्तन प्रतिबद्ध करें (`गिट कमिट -एम 'अद्भुत सुविधा जोड़ें'`) -4. शाखा में पुश करें (`गिट पुश ओरिजिन फीचर/अद्भुत-फीचर`) -5. एक पुल अनुरोध खोलें +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -विस्तृत दिशानिर्देशों के लिए [CONTRIBUTING.md](CONTRIBUTING.md) देखें।### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1994,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -**[9router](https://github.com/decolua/9router)**को**[decolua](https://github.com/decolua)**द्वारा विशेष धन्यवाद - मूल परियोजना जिसने इस फोर्क को प्रेरित किया। ओमनीरूट अतिरिक्त सुविधाओं, मल्टी-मोडल एपीआई और पूर्ण टाइपस्क्रिप्ट पुनर्लेखन के साथ उस अविश्वसनीय नींव पर आधारित है। +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -**[CLIPrxyAPI](https://github.com/router-for-me/CLIPrxyAPI)**को विशेष धन्यवाद - मूल गो कार्यान्वयन जिसने इस जावास्क्रिप्ट पोर्ट को प्रेरित किया।--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## लाइसेंस -एमआईटी लाइसेंस - विवरण के लिए [लाइसेंस](लाइसेंस) देखें।--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/hi/docs/ARCHITECTURE.md b/docs/i18n/hi/docs/ARCHITECTURE.md index 47de92cee1..adb69b5ed7 100644 --- a/docs/i18n/hi/docs/ARCHITECTURE.md +++ b/docs/i18n/hi/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_अंतिम अद्यतन: 2026-03-28_## Executive Summary -ओमनीरूट एक स्थानीय एआई रूटिंग गेटवे और नेक्स्ट.जेएस पर निर्मित डैशबोर्ड है। -यह एक एकल ओपनएआई-संगत एंडपॉइंट (`/v1/*`) प्रदान करता है और अनुवाद, फ़ॉलबैक, टोकन रिफ्रेश और उपयोग ट्रैकिंग के साथ कई अपस्ट्रीम प्रदाताओं के बीच ट्रैफ़िक को रूट करता है। -मुख्य क्षमताएं: +_Last updated: 2026-03-28_ -- सीएलआई/टूल्स के लिए ओपनएआई-संगत एपीआई सतह (28 प्रदाता) -- प्रदाता प्रारूपों में अनुरोध/प्रतिक्रिया अनुवाद -- मॉडल कॉम्बो फ़ॉलबैक (मल्टी-मॉडल अनुक्रम) -- खाता-स्तरीय फ़ॉलबैक (प्रति प्रदाता बहु-खाता) -- OAuth + एपीआई-कुंजी प्रदाता कनेक्शन प्रबंधन -- `/v1/embeddings` के माध्यम से एम्बेडिंग पीढ़ी (6 प्रदाता, 9 मॉडल) -- `/v1/images/पीढ़ी` के माध्यम से छवि निर्माण (4 प्रदाता, 9 मॉडल) -- तर्क मॉडल के लिए थिंक टैग पार्सिंग (`<थिंक>...`) सोचें -- सख्त ओपनएआई एसडीके संगतता के लिए प्रतिक्रिया स्वच्छता -- क्रॉस-प्रदाता अनुकूलता के लिए भूमिका सामान्यीकरण (डेवलपर→सिस्टम, सिस्टम→उपयोगकर्ता)। -- संरचित आउटपुट रूपांतरण (json_schema → जेमिनी रिस्पॉन्सस्कीमा) -- प्रदाताओं, चाबियाँ, उपनाम, कॉम्बो, सेटिंग्स, मूल्य निर्धारण के लिए स्थानीय दृढ़ता -- उपयोग/लागत ट्रैकिंग और अनुरोध लॉगिंग -- मल्टी-डिवाइस/स्टेट सिंक के लिए वैकल्पिक क्लाउड सिंक -- एपीआई एक्सेस नियंत्रण के लिए आईपी अनुमति सूची/ब्लॉकलिस्ट -- सोच बजट प्रबंधन (पासथ्रू/ऑटो/कस्टम/अनुकूली) -- वैश्विक प्रणाली शीघ्र इंजेक्शन -- सत्र ट्रैकिंग और फ़िंगरप्रिंटिंग -- प्रदाता-विशिष्ट प्रोफाइल के साथ प्रति-खाता बढ़ी हुई दर सीमित करना -- प्रदाता लचीलेपन के लिए सर्किट ब्रेकर पैटर्न -- म्यूटेक्स लॉकिंग के साथ एंटी-थंडरिंग झुंड सुरक्षा -- हस्ताक्षर-आधारित अनुरोध डिडुप्लीकेशन कैश -- डोमेन परत: मॉडल उपलब्धता, लागत नियम, फ़ॉलबैक नीति, लॉकआउट नीति -- डोमेन स्थिति दृढ़ता (फ़ॉलबैक, बजट, लॉकआउट, सर्किट ब्रेकर के लिए SQLite राइट-थ्रू कैश) -- केंद्रीकृत अनुरोध मूल्यांकन के लिए नीति इंजन (लॉकआउट → बजट → फ़ॉलबैक) -- p50/p95/p99 विलंबता एकत्रीकरण के साथ टेलीमेट्री का अनुरोध करें -- एंड-टू-एंड ट्रेसिंग के लिए सहसंबंध आईडी (एक्स-रिक्वेस्ट-आईडी)। -- एपीआई कुंजी के अनुसार ऑप्ट-आउट के साथ अनुपालन ऑडिट लॉगिंग -- एलएलएम गुणवत्ता आश्वासन के लिए इवल फ्रेमवर्क -- वास्तविक समय सर्किट ब्रेकर स्थिति के साथ लचीलापन यूआई डैशबोर्ड -- मॉड्यूलर OAuth प्रदाता (`src/lib/oauth/providers/` के अंतर्गत 12 व्यक्तिगत मॉड्यूल) +## Executive Summary -प्राथमिक रनटाइम मॉडल: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- `src/app/api/*` के अंतर्गत Next.js ऐप रूट डैशबोर्ड एपीआई और संगतता एपीआई दोनों को लागू करते हैं -- `src/sse/*` + `open-sse/*` में एक साझा SSE/रूटिंग कोर प्रदाता निष्पादन, अनुवाद, स्ट्रीमिंग, फ़ॉलबैक और उपयोग को संभालता है## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- स्थानीय गेटवे रनटाइम -- डैशबोर्ड प्रबंधन एपीआई -- प्रदाता प्रमाणीकरण और टोकन ताज़ा करें -- अनुवाद और एसएसई स्ट्रीमिंग का अनुरोध करें -- स्थानीय स्थिति + उपयोग की दृढ़ता -- वैकल्पिक क्लाउड सिंक ऑर्केस्ट्रेशन### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- `NEXT_PUBLIC_CLOUD_URL` के पीछे क्लाउड सेवा कार्यान्वयन -- स्थानीय प्रक्रिया के बाहर प्रदाता एसएलए/नियंत्रण विमान -- बाहरी सीएलआई बायनेरिज़ स्वयं (क्लाउड सीएलआई, कोडेक्स सीएलआई, आदि)## Dashboard Surface (Current) +### Out of Scope -`src/app/(डैशबोर्ड)/डैशबोर्ड/` के अंतर्गत मुख्य पृष्ठ: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/डैशबोर्ड` - त्वरित शुरुआत + प्रदाता अवलोकन -- `/डैशबोर्ड/एंडपॉइंट` - एंडपॉइंट प्रॉक्सी + एमसीपी + ए2ए + एपीआई एंडपॉइंट टैब -- `/डैशबोर्ड/प्रदाता` - प्रदाता कनेक्शन और क्रेडेंशियल -- `/डैशबोर्ड/कॉम्बोस` - कॉम्बो रणनीतियाँ, टेम्पलेट, मॉडल रूटिंग नियम -- `/डैशबोर्ड/लागत` - लागत एकत्रीकरण और मूल्य निर्धारण दृश्यता -- `/डैशबोर्ड/एनालिटिक्स` - उपयोग विश्लेषण और मूल्यांकन -- `/डैशबोर्ड/सीमाएँ` - कोटा/दर नियंत्रण -- `/डैशबोर्ड/क्ली-टूल्स` - सीएलआई ऑनबोर्डिंग, रनटाइम डिटेक्शन, कॉन्फिग जेनरेशन -- `/डैशबोर्ड/एजेंट` - पता चला एसीपी एजेंट + कस्टम एजेंट पंजीकरण -- `/डैशबोर्ड/मीडिया` - छवि/वीडियो/संगीत खेल का मैदान -- `/डैशबोर्ड/सर्च-टूल्स` - खोज प्रदाता परीक्षण और इतिहास -- `/डैशबोर्ड/स्वास्थ्य` - अपटाइम, सर्किट ब्रेकर, दर सीमा -- `/डैशबोर्ड/लॉग्स` - अनुरोध/प्रॉक्सी/ऑडिट/कंसोल लॉग -- `/डैशबोर्ड/सेटिंग्स` - सिस्टम सेटिंग्स टैब (सामान्य, रूटिंग, कॉम्बो डिफ़ॉल्ट, आदि) -- `/डैशबोर्ड/एपीआई-मैनेजर` - एपीआई कुंजी जीवनचक्र और मॉडल अनुमतियाँ## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -मुख्य निर्देशिकाएँ: +Main directories: -- अनुकूलता एपीआई के लिए `src/app/api/v1/*` और `src/app/api/v1beta/*` -- प्रबंधन/कॉन्फ़िगरेशन एपीआई के लिए `src/app/api/*` -- अगला `next.config.mjs` मैप `/v1/*` से `/api/v1/*` में फिर से लिखता है +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -महत्वपूर्ण अनुकूलता मार्ग: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` - इसमें `कस्टम: ट्रू` के साथ कस्टम मॉडल शामिल हैं -- `src/app/api/v1/embeddings/route.ts` - एम्बेडिंग जेनरेशन (6 प्रदाता) -- `src/app/api/v1/images/जेनरेशन/रूट.ts` - छवि निर्माण (एंटीग्रेविटी/नेबियस सहित 4+ प्रदाता) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` - प्रति-प्रदाता समर्पित चैट -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` - प्रति-प्रदाता समर्पित एम्बेडिंग -- `src/app/api/v1/providers/[provider]/images/nations/route.ts` - प्रति-प्रदाता समर्पित छवियां +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` - `src/app/api/v1beta/models/[...path]/route.ts` -प्रबंधन डोमेन: +Management domains: -- प्रामाणिक/सेटिंग्स: `src/app/api/auth/*`, `src/app/api/settings/*` -- प्रदाता/कनेक्शन: `src/app/api/प्रदाता*` -- प्रदाता नोड्स: `src/app/api/provider-nodes*` -- कस्टम मॉडल: `src/app/api/provider-models` (प्राप्त करें/पोस्ट करें/हटाएं) -- मॉडल कैटलॉग: `src/app/api/models/route.ts` (GET) -- प्रॉक्सी कॉन्फ़िगरेशन: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- कुंजी/उपनाम/कॉम्बोस/मूल्य निर्धारण: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- उपयोग: `src/app/api/usage/*` -- सिंक/क्लाउड: `src/app/api/sync/*`, `src/app/api/cloud/*` -- सीएलआई टूलींग सहायक: `src/app/api/cli-tools/*` -- आईपी फ़िल्टर: `src/app/api/settings/ip-filter` (प्राप्त/पुट) -- सोच बजट: `src/app/api/settings/thinking-budget` (प्राप्त/पुट) -- सिस्टम प्रॉम्प्ट: `src/app/api/settings/system-prompt` (GET/PUT) -- सत्र: `src/app/api/sessions` (प्राप्त करें) -- दर सीमा: `src/app/api/दर-सीमा` (प्राप्त करें) -- लचीलापन: `src/app/api/resilience` (GET/PATCH) - प्रदाता प्रोफाइल, सर्किट ब्रेकर, दर सीमा स्थिति -- लचीलापन रीसेट: `src/app/api/resilience/reset` (POST) - रीसेट ब्रेकर + कूलडाउन -- कैश आँकड़े: `src/app/api/cache/stats` (प्राप्त करें/हटाएँ) -- मॉडल उपलब्धता: `src/app/api/मॉडल/उपलब्धता` (प्राप्त करें/पोस्ट करें) -- टेलीमेट्री: `src/app/api/टेलीमेट्री/सारांश` (प्राप्त करें) -- बजट: `src/app/api/usage/budget` (प्राप्त करें/पोस्ट करें) -- फ़ॉलबैक चेन: `src/app/api/फ़ॉलबैक/चेन` (प्राप्त करें/पोस्ट करें/हटाएं) -- अनुपालन ऑडिट: `src/app/api/compliance/audit-log` (GET) -- इवल्स: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- नीतियां: `src/app/api/policies` (प्राप्त करें/पोस्ट करें)## 2) SSE + Translation Core +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -मुख्य प्रवाह मॉड्यूल: +## 2) SSE + Translation Core -- प्रविष्टि: `src/sse/handlers/chat.ts` -- कोर ऑर्केस्ट्रेशन: `open-sse/handlers/chatCore.ts` -- प्रदाता निष्पादन एडाप्टर: `ओपन-एसएसई/निष्पादक/*` -- प्रारूप पहचान/प्रदाता कॉन्फिगरेशन: `open-sse/services/provider.ts` -- मॉडल पार्स/रिज़ॉल्यूशन: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- खाता फ़ॉलबैक तर्क: `open-sse/services/accountFallback.ts` -- अनुवाद रजिस्ट्री: `open-sse/translator/index.ts` -- स्ट्रीम परिवर्तन: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- उपयोग निष्कर्षण/सामान्यीकरण: `open-sse/utils/usageTracking.ts` -- टैग पार्सर सोचें: `open-sse/utils/thinkTagParser.ts` -- एंबेडिंग हैंडलर: `open-sse/handlers/embeddings.ts` -- एंबेडिंग प्रदाता रजिस्ट्री: `open-sse/config/embeddingRegistry.ts` -- छवि निर्माण हैंडलर: `open-sse/handlers/imageGeneration.ts` -- छवि प्रदाता रजिस्ट्री: `open-sse/config/imageRegistry.ts` -- रिस्पॉन्स सेनिटाइजेशन: `ओपन-एसएसई/हैंडलर्स/रिस्पांससैनिटाइजर.टीएस` -- भूमिका सामान्यीकरण: `open-sse/services/roleNormalizer.ts` +Main flow modules: -सेवाएँ (व्यावसायिक तर्क): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- खाता चयन/स्कोरिंग: `open-sse/services/accountSelector.ts` -- संदर्भ जीवनचक्र प्रबंधन: `open-sse/services/contextManager.ts` -- आईपी फ़िल्टर प्रवर्तन: `open-sse/services/ipFilter.ts` -- सत्र ट्रैकिंग: `open-sse/services/sessionManager.ts` -- डुप्लिकेशन अनुरोध: `open-sse/services/signatureCache.ts` -- सिस्टम प्रॉम्प्ट इंजेक्शन: `open-sse/services/systemPrompt.ts` -- सोच बजट प्रबंधन: `open-sse/services/thinkingBudget.ts` -- वाइल्डकार्ड मॉडल रूटिंग: `open-sse/services/wildcardRouter.ts` -- दर सीमा प्रबंधन: `open-sse/services/rateLimitManager.ts` -- सर्किट ब्रेकर: `open-sse/services/circuitBreaker.ts` +Services (business logic): -डोमेन परत मॉड्यूल: +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- मॉडल उपलब्धता: `src/lib/domain/modelAvailability.ts` -- लागत नियम/बजट: `src/lib/domain/costRules.ts` -- फ़ॉलबैक नीति: `src/lib/domain/fallbackPolicy.ts` -- कॉम्बो रिज़ॉल्वर: `src/lib/domain/comboResolver.ts` -- लॉकआउट नीति: `src/lib/domain/lockoutPolicy.ts` -- नीति इंजन: `src/domain/policyEngine.ts` - केंद्रीकृत लॉकआउट → बजट → फ़ॉलबैक मूल्यांकन -- त्रुटि कोड कैटलॉग: `src/lib/domain/errorCodes.ts` -- अनुरोध आईडी: `src/lib/domain/requestId.ts` -- फ़ेच टाइमआउट: `src/lib/domain/fetchTimeout.ts` -- अनुरोध टेलीमेट्री: `src/lib/domain/requestTelemetry.ts` -- अनुपालन/ऑडिट: `src/lib/domain/compliance/index.ts` -- इवल रनर: `src/lib/domain/evalRunner.ts` -- डोमेन स्थिति दृढ़ता: `src/lib/db/domainState.ts` - फ़ॉलबैक चेन, बजट, लागत इतिहास, लॉकआउट स्थिति, सर्किट ब्रेकर के लिए SQLite CRUD +Domain layer modules: -OAuth प्रदाता मॉड्यूल (`src/lib/oauth/providers/` के अंतर्गत 12 व्यक्तिगत फ़ाइलें): +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -- रजिस्ट्री सूचकांक: `src/lib/oauth/providers/index.ts` -- व्यक्तिगत प्रदाता: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` -- पतला आवरण: `src/lib/oauth/providers.ts` - अलग-अलग मॉड्यूल से पुनः निर्यात## 3) Persistence Layer +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -प्राथमिक अवस्था DB (SQLite): +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -- कोर इन्फ्रा: `src/lib/db/core.ts` (बेहतर-sqlite3, माइग्रेशन, वाल) -- पुनः निर्यात पहलू: `src/lib/localDb.ts` (कॉलर्स के लिए पतली अनुकूलता परत) -- फ़ाइल: `${DATA_DIR}/storage.sqlite` (या `$XDG_CONFIG_HOME/omniroute/storage.sqlite` सेट होने पर, अन्यथा `~/.omniroute/storage.sqlite`) -- इकाइयां (टेबल + केवी नेमस्पेस): प्रोवाइडरकनेक्शन्स, प्रोवाइडरनोड्स, मॉडलएलियासेस, कॉम्बो, एपीआईकीज़, सेटिंग्स, मूल्य निर्धारण,**कस्टममॉडल**,**प्रॉक्सीकॉन्फिग**,**आईपीफिल्टर**,**थिंकिंगबजट**,**सिस्टमप्रॉम्प्ट** +## 3) Persistence Layer -उपयोग दृढ़ता: +Primary state DB (SQLite): -- मुखौटा: `src/lib/usageDb.ts` (`src/lib/usage/*` में विघटित मॉड्यूल) -- `storage.sqlite` में SQLite तालिकाएँ: `usage_history`, `call_logs`, `proxy_logs` +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** + +Usage persistence: + +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` - optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- मौजूद होने पर लीगेसी JSON फ़ाइलें स्टार्टअप माइग्रेशन द्वारा SQLite में माइग्रेट की जाती हैं +- legacy JSON files are migrated to SQLite by startup migrations when present -डोमेन स्थिति DB (SQLite): +Domain State DB (SQLite): -- `src/lib/db/domainState.ts` - डोमेन स्थिति के लिए CRUD संचालन -- टेबल्स (`src/lib/db/core.ts` में निर्मित): `domain_fallback_चेन्स`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- राइट-थ्रू कैश पैटर्न: इन-मेमोरी मैप्स रनटाइम पर आधिकारिक होते हैं; उत्परिवर्तन SQLite के साथ समकालिक रूप से लिखे जाते हैं; कोल्ड स्टार्ट पर राज्य को डीबी से बहाल किया जाता है## 4) Auth + Security Surfaces +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces - Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- एपीआई कुंजी निर्माण/सत्यापन: `src/shared/utils/apiKey.ts` -- प्रदाता रहस्य `providerConnections` प्रविष्टियों में बने रहे -- `open-sse/utils/proxyFetch.ts` (env vars) और `open-sse/utils/networkProxy.ts` (प्रति-प्रदाता या वैश्विक रूप से कॉन्फ़िगर करने योग्य) के माध्यम से आउटबाउंड प्रॉक्सी समर्थन## 5) Cloud Sync +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) -- शेड्यूलर init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- आवधिक कार्य: `src/shared/services/cloudSyncScheduler.ts` -- आवधिक कार्य: `src/shared/services/modelSyncScheduler.ts` -- नियंत्रण मार्ग: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -फ़ॉलबैक निर्णय स्थिति कोड और त्रुटि-संदेश अनुमानों का उपयोग करके `open-sse/services/accountFallback.ts` द्वारा संचालित होते हैं। कॉम्बो रूटिंग एक अतिरिक्त गार्ड जोड़ता है: प्रदाता-स्कोप्ड 400s जैसे अपस्ट्रीम सामग्री-ब्लॉक और भूमिका-सत्यापन विफलताओं को मॉडल-स्थानीय विफलताओं के रूप में माना जाता है ताकि बाद में कॉम्बो लक्ष्य अभी भी चल सकें।## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -लाइव ट्रैफ़िक के दौरान रिफ्रेश को निष्पादक `refreshCredentials()` के माध्यम से `open-sse/handlers/chatCore.ts` के अंदर निष्पादित किया जाता है।## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -503,12 +532,14 @@ erDiagram } ``` -भौतिक भंडारण फ़ाइलें: +Physical storage files: -- प्राथमिक रनटाइम DB: `${DATA_DIR}/storage.sqlite` -- अनुरोध लॉग लाइनें: `${DATA_DIR}/log.txt` (कॉम्पैट/डीबग आर्टिफैक्ट) -- संरचित कॉल पेलोड अभिलेखागार: `${DATA_DIR}/call_logs/` -- वैकल्पिक अनुवादक/अनुरोध डिबग सत्र: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -543,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: अनुकूलता एपीआई -- `src/app/api/v1/providers/[provider]/*`: प्रति-प्रदाता समर्पित मार्ग (चैट, एम्बेडिंग, चित्र) -- `src/app/api/providers*`: प्रदाता CRUD, सत्यापन, परीक्षण -- `src/app/api/provider-nodes*`: कस्टम संगत नोड प्रबंधन -- `src/app/api/provider-models`: कस्टम मॉडल प्रबंधन (CRUD) -- `src/app/api/models/route.ts`: मॉडल कैटलॉग एपीआई (उपनाम + कस्टम मॉडल) -- `src/app/api/oauth/*`: OAuth/डिवाइस-कोड प्रवाह -- `src/app/api/keys*`: स्थानीय एपीआई कुंजी जीवनचक्र -- `src/app/api/models/alias`: उपनाम प्रबंधन -- `src/app/api/combos*`: फ़ॉलबैक कॉम्बो प्रबंधन -- `src/app/api/pricing`: लागत गणना के लिए मूल्य निर्धारण ओवरराइड हो जाता है -- `src/app/api/settings/proxy`: प्रॉक्सी कॉन्फ़िगरेशन (प्राप्त/पुट/हटाएं) -- `src/app/api/settings/proxy/test`: आउटबाउंड प्रॉक्सी कनेक्टिविटी टेस्ट (POST) -- `src/app/api/usage/*`: एपीआई का उपयोग और लॉग -- `src/app/api/sync/*` + `src/app/api/cloud/*`: क्लाउड सिंक और क्लाउड-फेसिंग हेल्पर्स -- `src/app/api/cli-tools/*`: स्थानीय सीएलआई कॉन्फ़िगरेशन लेखक/चेकर्स -- `src/app/api/settings/ip-filter`: आईपी अनुमति सूची/ब्लॉकलिस्ट (प्राप्त/पुट) -- `src/app/api/settings/thinking-budget`: थिंकिंग टोकन बजट कॉन्फ़िगरेशन (GET/PUT) -- `src/app/api/settings/system-prompt`: ग्लोबल सिस्टम प्रॉम्प्ट (GET/PUT) -- `src/app/api/sessions`: सक्रिय सत्र सूची (GET) -- `src/app/api/rate-limits`: प्रति-खाता दर सीमा स्थिति (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: अनुरोध पार्स, कॉम्बो हैंडलिंग, खाता चयन लूप -- `ओपन-एसएसई/हैंडलर/चैटकोर.टीएस`: अनुवाद, निष्पादक प्रेषण, पुनः प्रयास/रीफ्रेश हैंडलिंग, स्ट्रीम सेटअप -- `ओपन-एसएसई/निष्पादक/*`: प्रदाता-विशिष्ट नेटवर्क और प्रारूप व्यवहार### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: अनुवादक रजिस्ट्री और ऑर्केस्ट्रेशन -- अनुवादकों से अनुरोध: `ओपन-एसएसई/अनुवादक/अनुरोध/*` -- प्रतिक्रिया अनुवादक: `ओपन-एसएसई/अनुवादक/प्रतिक्रिया/*` -- प्रारूप स्थिरांक: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: SQLite पर लगातार कॉन्फ़िगरेशन/स्थिति और डोमेन दृढ़ता -- `src/lib/localDb.ts`: डीबी मॉड्यूल के लिए अनुकूलता पुनः निर्यात -- `src/lib/usageDb.ts`: SQLite तालिकाओं के शीर्ष पर उपयोग इतिहास/कॉल लॉग मुखौटा## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -प्रत्येक प्रदाता के पास `BaseExecutor` (`open-sse/executors/base.ts` में) का विस्तार करने वाला एक विशेष निष्पादक होता है, जो URL निर्माण, हेडर निर्माण, घातीय बैकऑफ़ के साथ पुनः प्रयास, क्रेडेंशियल रिफ्रेश हुक और `execute()` ऑर्केस्ट्रेशन विधि प्रदान करता है। +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| निष्पादक | प्रदाता(ओं) | विशेष हैंडलिंग | -| --------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------- | -| `डिफ़ॉल्ट निष्पादक` | ओपनएआई, क्लाउड, जेमिनी, क्वेन, क्यूडर, ओपनराउटर, जीएलएम, किमी, मिनीमैक्स, डीपसीक, ग्रोक, एक्सएआई, मिस्ट्रल, पर्प्लेक्सिटी, टुगेदर, फायरवर्क्स, सेरेब्रा, कोहेरे, एनवीआईडीआईए | प्रति प्रदाता डायनामिक यूआरएल/हेडर कॉन्फिगरेशन | -| 'एंटीग्रेविटी एक्ज़ीक्यूटर' | गूगल एंटीग्रेविटी | कस्टम प्रोजेक्ट/सत्र आईडी, पुनः प्रयास करें-पार्सिंग के बाद | -| `कोडेक्स एक्ज़ीक्यूटर` ​​ | ओपनएआई कोडेक्स | सिस्टम निर्देश इंजेक्ट करता है, तर्क करने का प्रयास करता है | -| `कर्सर निष्पादक` | कर्सर आईडीई | कनेक्टआरपीसी प्रोटोकॉल, प्रोटोबफ एन्कोडिंग, चेकसम के माध्यम से हस्ताक्षर करने का अनुरोध | -| 'GithubExecutor' | गिटहब कोपायलट | कोपायलट टोकन ताज़ा करें, VSCode-नकल हेडर | -| `कीरो एक्ज़ीक्यूटर` ​​ | एडब्ल्यूएस कोडव्हिस्परर/किरो | एडब्ल्यूएस इवेंटस्ट्रीम बाइनरी प्रारूप → एसएसई रूपांतरण | -| `जेमिनीसीएलआईएक्सक्यूटर` ​​ | जेमिनी सीएलआई | Google OAuth टोकन ताज़ा चक्र | +### Persistence -अन्य सभी प्रदाता (कस्टम संगत नोड्स सहित) `DefaultExecutor` का उपयोग करते हैं।## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| प्रदाता | प्रारूप | प्रामाणिक | स्ट्रीम | नॉन-स्ट्रीम | टोकन ताज़ा करें | उपयोग एपीआई | -| --------------------- | -------------------- | ------------------------ | ----------------- | ----------- | --------------- | ------------------- | ------------------------------ | -| क्लाउड | क्लाउड | एपीआई कुंजी / OAuth | ✅ | ✅ | ✅ | ⚠️ केवल एडमिन | -| मिथुन | मिथुन | एपीआई कुंजी / OAuth | ✅ | ✅ | ✅ | ⚠️ क्लाउड कंसोल | -| जेमिनी सीएलआई | मिथुन-क्ली | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | -| प्रतिगुरुत्वाकर्षण | प्रतिगुरुत्वाकर्षण | OAuth | ✅ | ✅ | ✅ | ✅ पूर्ण कोटा एपीआई | -| ओपनएआई | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| कोडेक्स | openai-प्रतिक्रियाएं | OAuth | ✅ मजबूर | ❌ | ✅ | ✅ दर सीमा | -| गिटहब कोपायलट | ओपनाई | OAuth + सहपायलट टोकन | ✅ | ✅ | ✅ | ✅ कोटा स्नैपशॉट | -| कर्सर | कर्सर | कस्टम चेकसम | ✅ | ✅ | ❌ | ❌ | -| किरो | किरो | एडब्ल्यूएस एसएसओ ओआईडीसी | ✅ (इवेंटस्ट्रीम) | ❌ | ✅ | ✅ उपयोग सीमा | -| क्वेन | ओपनाई | OAuth | ✅ | ✅ | ✅ | ⚠️ प्रति अनुरोध | -| कोडर | ओपनाई | OAuth (बेसिक) | ✅ | ✅ | ✅ | ⚠️ प्रति अनुरोध | -| ओपनराउटर | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| जीएलएम/किमी/मिनीमैक्स | क्लाउड | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| डीपसीक | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| ग्रोक | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| एक्सएआई (ग्रोक) | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| मिस्ट्रल | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| उलझन | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| एक साथ एआई | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| आतिशबाजी एआई | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| सेरेब्रस | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| सहभागी | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| एनवीडिया एनआईएम | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -पता लगाए गए स्रोत प्रारूपों में शामिल हैं: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. -- `ओपनाई` -- `ओपनई-प्रतिक्रियाएँ` -- `क्लाउड` -- 'मिथुन' +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | -लक्ष्य प्रारूपों में शामिल हैं: +All other providers (including custom compatible nodes) use the `DefaultExecutor`. -- ओपनएआई चैट/प्रतिक्रियाएं - -क्लाउड -- मिथुन/मिथुन-सीएलआई/एंटीग्रेविटी लिफाफा -- किरो -- कर्सर +## Provider Compatibility Matrix -अनुवाद**हब प्रारूप के रूप में ओपनएआई**का उपयोग करते हैं - सभी रूपांतरण मध्यवर्ती के रूप में ओपनएआई से गुजरते हैं:``` +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: + +- `openai` +- `openai-responses` +- `claude` +- `gemini` + +Target formats include: + +- OpenAI chat/Responses +- Claude +- Gemini/Gemini-CLI/Antigravity envelope +- Kiro +- Cursor + +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -स्रोत पेलोड आकार और प्रदाता लक्ष्य प्रारूप के आधार पर अनुवादों का चयन गतिशील रूप से किया जाता है। +Additional processing layers in the translation pipeline: -अनुवाद पाइपलाइन में अतिरिक्त प्रसंस्करण परतें: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**प्रतिक्रिया स्वच्छता**- सख्त एसडीके अनुपालन सुनिश्चित करने के लिए ओपनएआई-प्रारूप प्रतिक्रियाओं (स्ट्रीमिंग और गैर-स्ट्रीमिंग दोनों) से गैर-मानक फ़ील्ड हटा देता है --**भूमिका सामान्यीकरण**- गैर-ओपनएआई लक्ष्यों के लिए `डेवलपर` → `सिस्टम` को रूपांतरित करता है; सिस्टम भूमिका को अस्वीकार करने वाले मॉडलों के लिए `सिस्टम` → `उपयोगकर्ता` को मर्ज करता है (जीएलएम, ईआरएनआईई) --**टैग निष्कर्षण के बारे में सोचें**- पार्स `<सोच>...` सामग्री से `reasoning_content` फ़ील्ड में ब्लॉक करता है --**संरचित आउटपुट**- OpenAI `response_format.json_schema` को जेमिनी के `responseMimeType` + `responseSchema` में परिवर्तित करता है## Supported API Endpoints +## Supported API Endpoints -| समापन बिंदु | प्रारूप | हैंडलर | -| -------------------------------------------------- | ------------------ | ---------------------------------------------------------------------------------- | -| `पोस्ट /v1/चैट/समापन` | ओपनएआई चैट | `src/sse/handlers/chat.ts` | -| `पोस्ट /v1/संदेश` | क्लाउड संदेश | वही हैंडलर (स्वतः पता चला) | -| `पोस्ट /v1/प्रतिक्रियाएँ` | ओपनएआई प्रतिक्रियाएँ | `open-sse/handlers/responsesHandler.ts` | -| `पोस्ट /v1/एम्बेडिंग` | ओपनएआई एंबेडिंग्स | `open-sse/handlers/embeddings.ts` | -| `प्राप्त करें /v1/एम्बेडिंग्स` | मॉडल सूची | एपीआई मार्ग | -| `पोस्ट /v1/छवियां/पीढ़ी` | OpenAI छवियाँ | `ओपन-एसएसई/हैंडलर/इमेजजेनरेशन.टीएस` | -| `प्राप्त करें /v1/छवियां/पीढ़ी` | मॉडल सूची | एपीआई मार्ग | -| `पोस्ट /v1/प्रदाता/{प्रदाता}/चैट/समापन` | ओपनएआई चैट | मॉडल सत्यापन के साथ प्रति-प्रदाता समर्पित | -| `पोस्ट /v1/प्रदाता/{प्रदाता}/एम्बेडिंग्स` | ओपनएआई एंबेडिंग्स | मॉडल सत्यापन के साथ प्रति-प्रदाता समर्पित | -| `पोस्ट /v1/प्रदाता/{प्रदाता}/छवियां/पीढ़ी` | OpenAI छवियाँ | मॉडल सत्यापन के साथ प्रति-प्रदाता समर्पित | -| `POST /v1/messages/count_tokens` | क्लाउड टोकन गिनती | एपीआई मार्ग | -| `प्राप्त करें /v1/मॉडल` | OpenAI मॉडल सूची | एपीआई मार्ग (चैट + एम्बेडिंग + छवि + कस्टम मॉडल) | -| `प्राप्त करें /एपीआई/मॉडल/कैटलॉग` | कैटलॉग | प्रदाता + प्रकार | द्वारा समूहीकृत सभी मॉडल -| `POST /v1beta/models/*:streamGenerateContent` | मिथुन राशि के जातक | एपीआई मार्ग | -| `प्राप्त/पुट/डिलीट /एपीआई/सेटिंग्स/प्रॉक्सी` | प्रॉक्सी कॉन्फिग | नेटवर्क प्रॉक्सी कॉन्फ़िगरेशन | -| `पोस्ट /एपीआई/सेटिंग्स/प्रॉक्सी/टेस्ट` | प्रॉक्सी कनेक्टिविटी | प्रॉक्सी स्वास्थ्य/कनेक्टिविटी परीक्षण समापन बिंदु | -| `प्राप्त करें/पोस्ट करें/हटाएं /एपीआई/प्रदाता-मॉडल` | प्रदाता मॉडल | प्रदाता मॉडल मेटाडेटा समर्थन कस्टम और प्रबंधित उपलब्ध मॉडल |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -बाईपास हैंडलर (`ओपन-एसएसई/यूटिल्स/बायपासहैंडलर.टीएस`) क्लाउड सीएलआई से ज्ञात "थ्रोअवे" अनुरोधों को रोकता है - वार्मअप पिंग, शीर्षक निष्कर्षण और टोकन गिनती - और अपस्ट्रीम प्रदाता टोकन का उपभोग किए बिना एक**नकली प्रतिक्रिया**देता है। यह तभी ट्रिगर होता है जब `User-Agent` में `claude-cli` होता है।## Request Logger Pipeline +## Bypass Handler -अनुरोध लकड़हारा (`open-sse/utils/requestLogger.ts`) एक 7-चरण डीबग लॉगिंग पाइपलाइन प्रदान करता है, जो डिफ़ॉल्ट रूप से अक्षम है, `ENABLE_REQUEST_LOGS=true` के माध्यम से सक्षम है:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -प्रत्येक अनुरोध सत्र के लिए फ़ाइलें `/logs//` पर लिखी जाती हैं।## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- क्षणिक/दर/प्रामाणिक त्रुटियों पर प्रदाता खाता ठंडा हो गया -- अनुरोध विफल होने से पहले खाता फ़ॉलबैक -- वर्तमान मॉडल/प्रदाता पथ समाप्त होने पर कॉम्बो मॉडल फ़ॉलबैक## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- ताज़ा करने योग्य प्रदाताओं के लिए पुनः प्रयास के साथ पूर्व-जांच और ताज़ा करें -- कोर पथ में ताज़ा प्रयास के बाद 401/403 पुनः प्रयास करें## 3) Stream Safety +## 2) Token Expiry -- डिस्कनेक्ट-अवेयर स्ट्रीम नियंत्रक -- एंड-ऑफ-स्ट्रीम फ्लश और `[DONE]` हैंडलिंग के साथ अनुवाद स्ट्रीम -- प्रदाता उपयोग मेटाडेटा अनुपलब्ध होने पर उपयोग अनुमान फ़ॉलबैक## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- समन्वयन त्रुटियाँ सामने आती हैं लेकिन स्थानीय रनटाइम जारी रहता है -- शेड्यूलर में पुनः प्रयास-सक्षम तर्क है, लेकिन आवधिक निष्पादन वर्तमान में डिफ़ॉल्ट रूप से एकल-प्रयास सिंक को कॉल करता है## 5) Data Integrity +## 3) Stream Safety -- स्टार्टअप पर SQLite स्कीमा माइग्रेशन और ऑटो-अपग्रेड हुक -- लीगेसी JSON → SQLite माइग्रेशन संगतता पथ## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -रनटाइम दृश्यता स्रोत: +## 4) Cloud Sync Degradation -- कंसोल `src/sse/utils/logger.ts` से लॉग करता है -- SQLite में प्रति-अनुरोध उपयोग समुच्चय (`use_history`, `call_logs`, `proxy_logs`) -- जब `settings.detailed_logs_enabled=true` होता है तो SQLite (`request_detail_logs`) में चार चरण वाला विस्तृत पेलोड कैप्चर होता है -- पाठ्य अनुरोध स्थिति लॉग इन `log.txt` (वैकल्पिक/कॉम्पैट) -- `ENABLE_REQUEST_LOGS=true` होने पर `लॉग/` के अंतर्गत वैकल्पिक गहन अनुरोध/अनुवाद लॉग -- यूआई खपत के लिए डैशबोर्ड उपयोग समापन बिंदु (`/api/usage/*`)। +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -विस्तृत अनुरोध पेलोड कैप्चर प्रति रूटेड कॉल को चार JSON पेलोड चरणों तक संग्रहीत करता है: +## 5) Data Integrity -- ग्राहक से प्राप्त कच्चा अनुरोध -- अनुवादित अनुरोध वास्तव में अपस्ट्रीम भेजा गया -- प्रदाता प्रतिक्रिया JSON के रूप में पुनर्निर्मित; स्ट्रीम की गई प्रतिक्रियाओं को अंतिम सारांश और स्ट्रीम मेटाडेटा में संकलित किया जाता है -- ओम्निरूट द्वारा लौटाई गई अंतिम ग्राहक प्रतिक्रिया; स्ट्रीम की गई प्रतिक्रियाएँ उसी संक्षिप्त सारांश रूप में संग्रहीत की जाती हैं## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- JWT सीक्रेट (`JWT_SECRET`) डैशबोर्ड सत्र कुकी सत्यापन/हस्ताक्षर को सुरक्षित करता है -- प्रारंभिक पासवर्ड बूटस्ट्रैप (`INITIAL_PASSWORD`) को प्रथम-रन प्रावधान के लिए स्पष्ट रूप से कॉन्फ़िगर किया जाना चाहिए -- एपीआई कुंजी HMAC सीक्रेट (`API_KEY_SECRET`) उत्पन्न स्थानीय एपीआई कुंजी प्रारूप को सुरक्षित करता है -- प्रदाता रहस्य (एपीआई कुंजी/टोकन) स्थानीय डीबी में बने रहते हैं और उन्हें फ़ाइल सिस्टम स्तर पर संरक्षित किया जाना चाहिए -- क्लाउड सिंक एंडपॉइंट एपीआई कुंजी ऑथ + मशीन आईडी सेमेन्टिक्स पर निर्भर करते हैं## Environment and Runtime Matrix +## Observability and Operational Signals -कोड द्वारा सक्रिय रूप से उपयोग किए जाने वाले पर्यावरण चर: +Runtime visibility sources: -- ऐप/ऑथ: `JWT_SECRET`, `INITIAL_PASSWORD` -- भंडारण: `DATA_DIR` -- संगत नोड व्यवहार: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- वैकल्पिक स्टोरेज बेस ओवरराइड (Linux/macOS जब `DATA_DIR` अनसेट होता है): `XDG_CONFIG_HOME` -- सुरक्षा हैशिंग: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- लॉगिंग: `ENABLE_REQUEST_LOGS` -- सिंक/क्लाउड यूआरएल: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- आउटबाउंड प्रॉक्सी: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` और लोअरकेस वेरिएंट -- SOCKS5 फ़ीचर फ़्लैग: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- प्लेटफ़ॉर्म/रनटाइम सहायक (ऐप-विशिष्ट कॉन्फ़िगरेशन नहीं): `एप्लिकेशन डेटा`, `NODE_ENV`, `पोर्ट`, `होस्टनाम`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` और `localDb` लीगेसी फ़ाइल माइग्रेशन के साथ समान आधार निर्देशिका नीति (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) साझा करते हैं। -2. `/api/v1/route.ts` सिमेंटिक बहाव से बचने के लिए `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) द्वारा उपयोग किए जाने वाले समान एकीकृत कैटलॉग बिल्डर को सौंपता है। -3. अनुरोध लकड़हारा सक्षम होने पर पूर्ण हेडर/बॉडी लिखता है; लॉग निर्देशिका को संवेदनशील मानें। -4. क्लाउड व्यवहार सही `NEXT_PUBLIC_BASE_URL` और क्लाउड एंडपॉइंट रीचैबिलिटी पर निर्भर करता है। -5. `ओपन-एसएसई/` निर्देशिका को `@omniroute/ओपन-एसएसई`**एनपीएम वर्कस्पेस पैकेज**के रूप में प्रकाशित किया गया है। स्रोत कोड इसे `@omniroute/open-sse/...` (Next.js `transpilePackages` द्वारा हल) के माध्यम से आयात करता है। इस दस्तावेज़ में फ़ाइल पथ अभी भी स्थिरता के लिए निर्देशिका नाम `open-sse/` का उपयोग करते हैं। -6. डैशबोर्ड में चार्ट सुलभ, इंटरैक्टिव एनालिटिक्स विज़ुअलाइज़ेशन (मॉडल उपयोग बार चार्ट, सफलता दर के साथ प्रदाता ब्रेकडाउन टेबल) के लिए**रिचार्ट्स**(एसवीजी-आधारित) का उपयोग करते हैं। -7. E2E परीक्षण**Playwright**(`test/e2e/`) का उपयोग करते हैं, `npm run test:e2e` के माध्यम से चलते हैं। यूनिट परीक्षण**नोड.जेएस टेस्ट रनर**(`टेस्ट/यूनिट/`) का उपयोग करते हैं, जो `एनपीएम रन टेस्ट: यूनिट` के माध्यम से चलते हैं। `src/` के अंतर्गत स्रोत कोड**टाइपस्क्रिप्ट**(`.ts`/`.tsx`) है; `ओपन-एसएसई/` कार्यक्षेत्र जावास्क्रिप्ट (`.जेएस`) बना हुआ है। -8. सेटिंग्स पृष्ठ को 5 टैब में व्यवस्थित किया गया है: सुरक्षा, रूटिंग (6 वैश्विक रणनीतियाँ: भरण-प्रथम, राउंड-रॉबिन, पी2सी, यादृच्छिक, कम से कम उपयोग किया गया, लागत-अनुकूलित), लचीलापन (संपादन योग्य दर सीमा, सर्किट ब्रेकर, नीतियां), एआई (सोच बजट, सिस्टम प्रॉम्प्ट, प्रॉम्प्ट कैश), उन्नत (प्रॉक्सी)।## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- स्रोत से निर्माण: `एनपीएम रन बिल्ड` -- डॉकर छवि बनाएँ: `docker build -t omniroute।` -- सेवा प्रारंभ करें और सत्यापित करें: -- `प्राप्त करें /एपीआई/सेटिंग्स` -- `प्राप्त करें /api/v1/मॉडल` -- जब `PORT=20128` हो तो CLI लक्ष्य आधार URL `http://:20128/v1` होना चाहिए +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` +- `GET /api/v1/models` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/hi/docs/FEATURES.md b/docs/i18n/hi/docs/FEATURES.md index 7cce5a61c0..46627da387 100644 --- a/docs/i18n/hi/docs/FEATURES.md +++ b/docs/i18n/hi/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -ओमनीरूट डैशबोर्ड के प्रत्येक अनुभाग के लिए विज़ुअल गाइड।--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -एआई प्रदाता कनेक्शन प्रबंधित करें: OAuth प्रदाता (क्लाउड कोड, कोडेक्स, जेमिनी सीएलआई), एपीआई कुंजी प्रदाता (ग्रोक, डीपसीक, ओपनराउटर), और मुफ्त प्रदाता (क्यूडर, क्वेन, किरो)। किरो खातों में क्रेडिट बैलेंस ट्रैकिंग - शेष क्रेडिट, कुल भत्ता और डैशबोर्ड → उपयोग में दिखाई देने वाली नवीनीकरण तिथि शामिल है।![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -6 रणनीतियों के साथ मॉडल रूटिंग कॉम्बो बनाएं: प्राथमिकता, भारित, राउंड-रॉबिन, यादृच्छिक, कम से कम उपयोग किया गया और लागत-अनुकूलित। प्रत्येक कॉम्बो स्वचालित फ़ॉलबैक के साथ कई मॉडलों को श्रृंखलाबद्ध करता है और इसमें त्वरित टेम्पलेट और तत्परता जांच शामिल होती है।![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -टोकन खपत, लागत अनुमान, गतिविधि हीटमैप, साप्ताहिक वितरण चार्ट और प्रति-प्रदाता विश्लेषण के साथ व्यापक उपयोग विश्लेषण।![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -वास्तविक समय की निगरानी: अपटाइम, मेमोरी, संस्करण, विलंबता प्रतिशत (p50/p95/p99), कैश आँकड़े, और प्रदाता सर्किट ब्रेकर स्थिति।![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -एपीआई अनुवादों को डीबग करने के लिए चार मोड:**प्लेग्राउंड**(फॉर्मेट कनवर्टर),**चैट टेस्टर**(लाइव अनुरोध),**टेस्ट बेंच**(बैच टेस्ट), और**लाइव मॉनिटर**(रियल-टाइम स्ट्रीम)।![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -किसी भी मॉडल का सीधे डैशबोर्ड से परीक्षण करें। प्रदाता, मॉडल और समापन बिंदु का चयन करें, मोनाको संपादक के साथ संकेत लिखें, वास्तविक समय में प्रतिक्रियाओं को स्ट्रीम करें, मध्य-धारा को निरस्त करें, और समय मेट्रिक्स देखें।--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -संपूर्ण डैशबोर्ड के लिए अनुकूलन योग्य रंग थीम। 7 पूर्व निर्धारित रंगों (कोरल, नीला, लाल, हरा, बैंगनी, नारंगी, सियान) में से चुनें या कोई भी हेक्स रंग चुनकर एक कस्टम थीम बनाएं। प्रकाश, अंधेरा और सिस्टम मोड का समर्थन करता है।--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -टैब के साथ व्यापक सेटिंग पैनल: +Comprehensive settings panel with tabs: --**सामान्य**- सिस्टम स्टोरेज, बैकअप प्रबंधन (निर्यात/आयात डेटाबेस) -**प्रकटन**- थीम चयनकर्ता (गहरा/प्रकाश/सिस्टम), रंग थीम प्रीसेट और कस्टम रंग, स्वास्थ्य लॉग दृश्यता, साइडबार आइटम दृश्यता नियंत्रण -**सुरक्षा**- एपीआई एंडपॉइंट सुरक्षा, कस्टम प्रदाता अवरोधन, आईपी फ़िल्टरिंग, सत्र जानकारी -**रूटिंग**- मॉडल उपनाम, पृष्ठभूमि कार्य गिरावट -**लचीलापन**- दर सीमा दृढ़ता, सर्किट ब्रेकर ट्यूनिंग, प्रतिबंधित खातों को स्वचालित रूप से अक्षम करें, प्रदाता समाप्ति की निगरानी -**उन्नत**- कॉन्फ़िगरेशन ओवरराइड, कॉन्फ़िगरेशन ऑडिट ट्रेल, फ़ॉलबैक डिग्रेडेशन मोड![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -एआई कोडिंग टूल के लिए एक-क्लिक कॉन्फ़िगरेशन: क्लाउड कोड, कोडेक्स सीएलआई, जेमिनी सीएलआई, ओपनक्लाव, किलो कोड, एंटीग्रेविटी, क्लाइन, कंटिन्यू, कर्सर और फैक्ट्री ड्रॉयड। सुविधाएँ स्वचालित कॉन्फ़िगरेशन लागू/रीसेट, कनेक्शन प्रोफ़ाइल और मॉडल मैपिंग।![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -सीएलआई एजेंटों की खोज और प्रबंधन के लिए डैशबोर्ड। 14 अंतर्निहित एजेंटों (कोडेक्स, क्लाउड, गूज़, जेमिनी सीएलआई, ओपनक्लाव, एडर, ओपनकोड, क्लाइन, क्वेन कोड, फोर्जकोड, अमेज़ॅन क्यू, ओपन इंटरप्रेटर, कर्सर सीएलआई, वार्प) का ग्रिड दिखाता है: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**इंस्टॉलेशन स्थिति**- संस्करण पहचान के साथ स्थापित / नहीं मिला -**Protocol badges**— stdio, HTTP, etc. -**कस्टम एजेंट**- फॉर्म के माध्यम से किसी भी सीएलआई टूल को पंजीकृत करें (नाम, बाइनरी, वर्जन कमांड, स्पॉन आर्ग्स) -**सीएलआई फ़िंगरप्रिंट मिलान**- मूल सीएलआई अनुरोध हस्ताक्षरों से मिलान करने के लिए प्रति-प्रदाता टॉगल करता है, प्रॉक्सी आईपी को संरक्षित करते हुए प्रतिबंध जोखिम को कम करता है--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -डैशबोर्ड से चित्र, वीडियो और संगीत उत्पन्न करें। OpenAI, xAI, टुगेदर, हाइपरबोलिक, SD WebUI, ComfyUI, AnimateDiff, स्टेबल ऑडियो ओपन और MusicGen को सपोर्ट करता है।--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -प्रदाता, मॉडल, खाता और एपीआई कुंजी द्वारा फ़िल्टरिंग के साथ वास्तविक समय अनुरोध लॉगिंग। स्थिति कोड, टोकन उपयोग, विलंबता और प्रतिक्रिया विवरण दिखाता है।![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -क्षमता विश्लेषण के साथ आपका एकीकृत एपीआई एंडपॉइंट: चैट पूर्णताएं, प्रतिक्रिया एपीआई, एंबेडिंग, छवि निर्माण, रीरैंकिंग, ऑडियो ट्रांसक्रिप्शन, टेक्स्ट-टू-स्पीच, मॉडरेशन और पंजीकृत एपीआई कुंजी। रिमोट एक्सेस के लिए क्लाउडफ्लेयर क्विक टनल इंटीग्रेशन और क्लाउड प्रॉक्सी सपोर्ट।![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -एपीआई कुंजी बनाएं, दायरा बढ़ाएं और निरस्त करें। प्रत्येक कुंजी को पूर्ण पहुंच या केवल-पढ़ने की अनुमति वाले विशिष्ट मॉडल/प्रदाताओं तक सीमित किया जा सकता है। उपयोग ट्रैकिंग के साथ विज़ुअल कुंजी प्रबंधन।--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -कार्रवाई प्रकार, अभिनेता, लक्ष्य, आईपी पता और टाइमस्टैम्प द्वारा फ़िल्टरिंग के साथ प्रशासनिक कार्रवाई ट्रैकिंग। पूर्ण सुरक्षा घटना इतिहास.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -विंडोज़, मैकओएस और लिनक्स के लिए नेटिव इलेक्ट्रॉन डेस्कटॉप ऐप। सिस्टम ट्रे एकीकरण, ऑफ़लाइन समर्थन, ऑटो-अपडेट और एक-क्लिक इंस्टॉल के साथ ओमनीरूट को एक स्टैंडअलोन एप्लिकेशन के रूप में चलाएं। +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -मुख्य विशेषताएं: +Key features: -- सर्वर तत्परता मतदान (कोल्ड स्टार्ट पर कोई खाली स्क्रीन नहीं) -- पोर्ट प्रबंधन के साथ सिस्टम ट्रे -- सामग्री सुरक्षा नीति -- सिंगल-इंस्टेंस लॉक -- पुनरारंभ पर स्वतः अद्यतन -- प्लेटफ़ॉर्म-सशर्त यूआई (मैकओएस ट्रैफिक लाइट, विंडोज़/लिनक्स डिफ़ॉल्ट टाइटलबार) -- कठोर इलेक्ट्रॉन बिल्ड पैकेजिंग - स्टैंडअलोन बंडल में सिम्लिंक्ड `नोड_मॉड्यूल` का पता लगाया जाता है और पैकेजिंग से पहले खारिज कर दिया जाता है, जिससे बिल्ड मशीन पर रनटाइम निर्भरता को रोका जा सकता है (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 संपूर्ण दस्तावेज़ीकरण के लिए [`electron/README.md`](../electron/README.md) देखें। +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/hi/docs/TROUBLESHOOTING.md b/docs/i18n/hi/docs/TROUBLESHOOTING.md index 5a582de15d..5fb25a9ebb 100644 --- a/docs/i18n/hi/docs/TROUBLESHOOTING.md +++ b/docs/i18n/hi/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -ओम्निरूट के लिए सामान्य समस्याएं और समाधान।--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| समस्या | समाधान | -| ------------------------------------- | ------------------------------------------------------------------------------- | --- | -| पहला लॉगिन काम नहीं कर रहा | `INITIAL_PASSWORD` को `.env` में सेट करें (कोई हार्डकोडेड डिफ़ॉल्ट नहीं) | -| गलत पोर्ट पर डैशबोर्ड खुलता है | `PORT=20128` और `NEXT_PUBLIC_BASE_URL=http://localhost:20128` सेट करें | -| `लॉग/` के अंतर्गत कोई अनुरोध लॉग नहीं | `ENABLE_REQUEST_LOGS=true` सेट करें | -| EACCES: अनुमति अस्वीकृत | `~/.omniroute` को ओवरराइड करने के लिए `DATA_DIR=/path/to/writable/dir` सेट करें | -| रूटिंग रणनीति सहेजी नहीं जा रही | v1.4.11+ पर अपडेट करें (सेटिंग्स दृढ़ता के लिए ज़ोड स्कीमा फिक्स) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**कारण:**प्रदाता कोटा समाप्त हो गया। +**Cause:** Provider quota exhausted. -**ठीक करें:** +**Fix:** -1. डैशबोर्ड कोटा ट्रैकर की जाँच करें -2. फ़ॉलबैक टियर वाले कॉम्बो का उपयोग करें -3. सस्ते/मुफ़्त स्तर पर स्विच करें### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**कारण:**सदस्यता कोटा समाप्त हो गया। +### Rate Limiting -**ठीक करें:** +**Cause:** Subscription quota exhausted. -- फ़ॉलबैक जोड़ें: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- सस्ते बैकअप के रूप में GLM/MiniMax का उपयोग करें### OAuth Token Expired +**Fix:** -ओम्निरूट स्वचालित रूप से टोकन ताज़ा करता है। यदि समस्याएँ बनी रहती हैं: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. डैशबोर्ड → प्रदाता → पुनः कनेक्ट करें -2. प्रदाता कनेक्शन हटाएं और पुनः जोड़ें--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. अपने चल रहे उदाहरण के लिए `BASE_URL` बिंदुओं को सत्यापित करें (उदाहरण के लिए, `http://localhost:20128`) -2. अपने क्लाउड एंडपॉइंट पर `CLOUD_URL` बिंदुओं को सत्यापित करें (उदाहरण के लिए, `https://omniroute.dev`) -3. `NEXT_PUBLIC_*` मानों को सर्वर-साइड मानों के साथ संरेखित रखें### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**लक्षण:**गैर-स्ट्रीमिंग कॉल के लिए क्लाउड एंडपॉइंट पर `अप्रत्याशित टोकन 'डी'...`। +### Cloud `stream=false` Returns 500 -**कारण:**अपस्ट्रीम एसएसई पेलोड लौटाता है जबकि ग्राहक JSON की अपेक्षा करता है। +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**समाधान:**क्लाउड डायरेक्ट कॉल के लिए `stream=true` का उपयोग करें। स्थानीय रनटाइम में SSE→JSON फ़ॉलबैक शामिल है।### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. स्थानीय डैशबोर्ड से एक नई कुंजी बनाएं (`/api/keys`) -2. क्लाउड सिंक चलाएँ: क्लाउड सक्षम करें → अभी सिंक करें -3. पुरानी/गैर-सिंक की गई कुंजियाँ अभी भी क्लाउड पर `401` लौटा सकती हैं--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. रनटाइम फ़ील्ड जांचें: `कर्ल http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. पोर्टेबल मोड के लिए: छवि लक्ष्य `रनर-सीएलआई` (बंडल सीएलआई) का उपयोग करें -3. होस्ट माउंट मोड के लिए: `CLI_EXTRA_PATHS` सेट करें और होस्ट बिन निर्देशिका को केवल पढ़ने के लिए माउंट करें -4. यदि `इंस्टॉल = सही` और `रनने योग्य = गलत`: बाइनरी पाया गया था लेकिन स्वास्थ्य जांच विफल रही### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. डैशबोर्ड → उपयोग में उपयोग के आँकड़े जाँचें -2. प्राथमिक मॉडल को जीएलएम/मिनीमैक्स पर स्विच करें -3. गैर-महत्वपूर्ण कार्यों के लिए फ्री टियर (मिथुन सीएलआई, क्यूडर) का उपयोग करें -4. प्रति एपीआई कुंजी लागत बजट निर्धारित करें: डैशबोर्ड → एपीआई कुंजी → बजट--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -अपनी `.env` फ़ाइल में `ENABLE_REQUEST_LOGS=true` सेट करें। लॉग `लॉग/` निर्देशिका के अंतर्गत दिखाई देते हैं।### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,98 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- मुख्य स्थिति: `${DATA_DIR}/storage.sqlite` (प्रदाता, कॉम्बो, उपनाम, कुंजियाँ, सेटिंग्स) -- उपयोग: `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) में SQLite टेबल + वैकल्पिक `${DATA_DIR}/log.txt` और `${DATA_DIR}/call_logs/` -- अनुरोध लॉग: `/logs/...` (जब `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -जब किसी प्रदाता का सर्किट ब्रेकर खुला होता है, तो कूलडाउन समाप्त होने तक अनुरोध अवरुद्ध हो जाते हैं। +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**ठीक करें:** +**Fix:** -1.**डैशबोर्ड → सेटिंग्स → लचीलापन**पर जाएं 2. प्रभावित प्रदाता के लिए सर्किट ब्रेकर कार्ड की जाँच करें 3. सभी ब्रेकर साफ़ करने के लिए**रीसेट ऑल**पर क्लिक करें, या कूलडाउन समाप्त होने तक प्रतीक्षा करें 4. रीसेट करने से पहले सत्यापित करें कि प्रदाता वास्तव में उपलब्ध है### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -यदि कोई प्रदाता बार-बार खुली स्थिति में प्रवेश करता है: +### Provider keeps tripping the circuit breaker -1. विफलता पैटर्न के लिए**डैशबोर्ड → स्वास्थ्य → प्रदाता स्वास्थ्य**की जाँच करें 2.**सेटिंग्स → लचीलापन → प्रदाता प्रोफाइल**पर जाएं और विफलता सीमा बढ़ाएं -2. जांचें कि क्या प्रदाता ने एपीआई सीमाएं बदल दी हैं या पुनः प्रमाणीकरण की आवश्यकता है -3. विलंबता टेलीमेट्री की समीक्षा करें - उच्च विलंबता टाइमआउट-आधारित विफलताओं का कारण बन सकती है--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- सुनिश्चित करें कि आप सही उपसर्ग का उपयोग कर रहे हैं: `डीपग्राम/नोवा-3` या `असेंबलीई/बेस्ट` -- सत्यापित करें कि प्रदाता**डैशबोर्ड → प्रदाता**में जुड़ा हुआ है### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- समर्थित ऑडियो प्रारूप जांचें: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- सत्यापित करें कि फ़ाइल का आकार प्रदाता सीमा के भीतर है (आमतौर पर <25MB) -- प्रदाता कार्ड में प्रदाता एपीआई कुंजी वैधता की जांच करें--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -प्रारूप अनुवाद समस्याओं को डीबग करने के लिए**डैशबोर्ड → अनुवादक**का उपयोग करें: +Use **Dashboard → Translator** to debug format translation issues: -| मोड | कब उपयोग करें | -| ---------------- | ------------------------------------------------------------------------------------------------------------- | ------------------------ | -| **खेल का मैदान** | इनपुट/आउटपुट स्वरूपों की साथ-साथ तुलना करें - यह कैसे अनुवादित होता है यह देखने के लिए एक असफल अनुरोध चिपकाएँ | -| **Chat Tester** | लाइव संदेश भेजें और हेडर सहित पूर्ण अनुरोध/प्रतिक्रिया पेलोड का निरीक्षण करें | -| **टेस्ट बेंच** | यह पता लगाने के लिए कि कौन से अनुवाद टूटे हुए हैं, सभी प्रारूप संयोजनों में बैच परीक्षण चलाएँ | -| **लाइव मॉनिटर** | रुक-रुक कर होने वाली अनुवाद समस्याओं को पकड़ने के लिए वास्तविक समय अनुरोध प्रवाह देखें | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**सोच टैग दिखाई नहीं दे रहे हैं**- जांचें कि क्या लक्ष्य प्रदाता सोच और सोच बजट सेटिंग का समर्थन करता है -**टूल कॉल ड्रॉपिंग**— कुछ प्रारूप अनुवाद असमर्थित फ़ील्ड को हटा सकते हैं; खेल का मैदान मोड में सत्यापित करें -**सिस्टम प्रॉम्प्ट गायब**— क्लाउड और जेमिनी हैंडल सिस्टम प्रॉम्प्ट अलग-अलग होते हैं; अनुवाद आउटपुट की जाँच करें -**एसडीके ऑब्जेक्ट के बजाय कच्ची स्ट्रिंग लौटाता है**- v1.1.0 में फिक्स्ड: रिस्पॉन्स सैनिटाइज़र अब गैर-मानक फ़ील्ड (`x_groq`, `usage_breakdown`, आदि) को हटा देता है जो OpenAI SDK पायडेंटिक सत्यापन विफलताओं का कारण बनता है -**GLM/ERNIE `सिस्टम' भूमिका को अस्वीकार करता है**- v1.1.0 में फिक्स्ड: रोल नॉर्मलाइज़र स्वचालित रूप से असंगत मॉडल के लिए सिस्टम संदेशों को उपयोगकर्ता संदेशों में मर्ज कर देता है -**'डेवलपर' की भूमिका पहचानी नहीं गई**- v1.1.0 में फिक्स्ड: गैर-ओपनएआई प्रदाताओं के लिए स्वचालित रूप से `सिस्टम' में कनवर्ट किया गया --**`json_schema`जेमिनी के साथ काम नहीं कर रहा है**- v1.1.0 में फिक्स्ड:`response_format`को अब जेमिनी के`responseMimeType`+`responseSchema` में बदल दिया गया है--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- ऑटो दर-सीमा केवल एपीआई कुंजी प्रदाताओं पर लागू होती है (OAuth/सदस्यता पर नहीं) -- सत्यापित करें**सेटिंग्स → लचीलापन → प्रदाता प्रोफाइल**में ऑटो-दर-सीमा सक्षम है -- जांचें कि क्या प्रदाता `429` स्टेटस कोड या `रीट्री-आफ्टर` हेडर लौटाता है### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -प्रदाता प्रोफ़ाइल इन सेटिंग्स का समर्थन करती हैं: +### Tuning exponential backoff --**आधार विलंब**— पहली विफलता के बाद प्रारंभिक प्रतीक्षा समय (डिफ़ॉल्ट: 1 सेकंड) -**अधिकतम विलंब**— अधिकतम प्रतीक्षा समय सीमा (डिफ़ॉल्ट: 30s) -**गुणक**- लगातार विफलता के बाद विलंब को कितना बढ़ाया जाए (डिफ़ॉल्ट: 2x)### Anti-thundering herd +Provider profiles support these settings: -जब कई समवर्ती अनुरोध एक दर-सीमित प्रदाता से टकराते हैं, तो ओमनीरूट अनुरोधों को क्रमबद्ध करने और कैस्केडिंग विफलताओं को रोकने के लिए म्यूटेक्स + ऑटो रेट-लिमिटिंग का उपयोग करता है। यह एपीआई कुंजी प्रदाताओं के लिए स्वचालित है।--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -कुछ ओमनीरूट उपयोगकर्ता गेटवे को RAG या एजेंट स्टैक के सामने रखते हैं। उन सेटअपों में एक अजीब पैटर्न देखना आम है: ओम्नीरूट स्वस्थ दिखता है (प्रदाता ऊपर, रूटिंग प्रोफाइल ठीक, कोई दर सीमा अलर्ट नहीं) लेकिन अंतिम उत्तर अभी भी गलत है। +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -व्यवहार में ये घटनाएं आम तौर पर डाउनस्ट्रीम आरएजी पाइपलाइन से आती हैं, गेटवे से नहीं। +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -यदि आप उन विफलताओं का वर्णन करने के लिए एक साझा शब्दावली चाहते हैं तो आप डब्लूएफजीवाई प्रॉब्लममैप का उपयोग कर सकते हैं, एक बाहरी एमआईटी लाइसेंस टेक्स्ट संसाधन जो सोलह आवर्ती आरएजी / एलएलएम विफलता पैटर्न को परिभाषित करता है। उच्च स्तर पर इसमें शामिल हैं: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- पुनर्प्राप्ति बहाव और टूटी हुई संदर्भ सीमाएँ -- खाली या बासी इंडेक्स और वेक्टर स्टोर -- एम्बेडिंग बनाम सिमेंटिक बेमेल -- शीघ्र असेंबली और संदर्भ विंडो समस्याएँ -- तर्क पतन और अतिआत्मविश्वासपूर्ण उत्तर -- लंबी श्रृंखला और एजेंट समन्वय विफलताएँ -- मल्टी एजेंट मेमोरी और रोल ड्रिफ्ट -- परिनियोजन और बूटस्ट्रैप ऑर्डरिंग समस्याएं +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -विचार सरल है: +The idea is simple: -1. जब आप किसी खराब प्रतिक्रिया की जांच करते हैं, तो कैप्चर करें: - - उपयोगकर्ता कार्य और अनुरोध - - ओमनीरूट में रूट या प्रदाता कॉम्बो - - डाउनस्ट्रीम में उपयोग किया गया कोई भी RAG संदर्भ (पुनर्प्राप्त दस्तावेज़, टूल कॉल, आदि) -2. घटना को एक या दो WFGY समस्या मानचित्र संख्याओं ('नंबर 1' ... 'नंबर 16') पर मैप करें। -3. नंबर को अपने डैशबोर्ड, रनबुक, या घटना ट्रैकर में ओमनीरूट लॉग के बगल में संग्रहीत करें। -4. यह तय करने के लिए संबंधित WFGY पृष्ठ का उपयोग करें कि आपको अपने RAG स्टैक, रिट्रीवर या रूटिंग रणनीति को बदलने की आवश्यकता है या नहीं। +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -पूर्ण पाठ और ठोस व्यंजन यहां उपलब्ध हैं (एमआईटी लाइसेंस, केवल पाठ): +Full text and concrete recipes live here (MIT license, text only): -[डब्ल्यूएफजीवाई प्रॉब्लममैप रीडमी](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) +[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -यदि आप ओमनीरूट के पीछे आरएजी या एजेंट पाइपलाइन नहीं चलाते हैं तो आप इस अनुभाग को अनदेखा कर सकते हैं।--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**गिटहब मुद्दे**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**आर्किटेक्चर**: आंतरिक विवरण के लिए [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) देखें -**एपीआई संदर्भ**: सभी समापन बिंदुओं के लिए [`docs/API_REFERENCE.md`](API_REFERENCE.md) देखें -**स्वास्थ्य डैशबोर्ड**: वास्तविक समय प्रणाली की स्थिति के लिए**डैशबोर्ड → स्वास्थ्य**जांचें -**अनुवादक**: प्रारूप संबंधी समस्याओं को डीबग करने के लिए**डैशबोर्ड → अनुवादक**का उपयोग करें +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/hi/llm.txt b/docs/i18n/hi/llm.txt new file mode 100644 index 0000000000..7a4c81f081 --- /dev/null +++ b/docs/i18n/hi/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (हिन्दी) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## अवलोकन + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### सुरक्षा +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/hu/README.md b/docs/i18n/hu/README.md index 30edc92590..88e3f44d8e 100644 --- a/docs/i18n/hu/README.md +++ b/docs/i18n/hu/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Az univerzális API-proxy – egy végpont, több mint 60 szolgáltató, nulla állásidő. Most**MCP-kiszolgálóval (25 eszköz)**,**A2A protokollal**,**memória-/készségrendszerekkel**és**Electron Desktop alkalmazással**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Csevegés befejezése • Beágyazások • Képgenerálás • Videó • Zene • Hang • Újrarangsorolás •**Webes keresés**• MCP-szerver • A2A protokoll • 100% TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Az univerzális API-proxy – egy végpont, több mint 60 szolgáltató, nulla [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Webhely](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Funkciók](#-key-features) • [📖 Dokumentumok](#-dokumentáció) • [💰 Árak](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Elérhető:**🇺🇸 [angol](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugália)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Szlovénia](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [filippínó](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,555 +60,629 @@ _Az univerzális API-proxy – egy végpont, több mint 60 szolgáltató, nulla ## 📸 Dashboard Preview - +
+Click to see dashboard screenshots -Kattintson ide az irányítópult képernyőképeinek megtekintéséhez +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | -| Oldal | Képernyőkép | -| --------------------- | -------------------------------------------------- | ---------- | -| **Szolgáltatók** | ![Szolgáltatók](docs/screenshots/01-providers.png) | -| **Kombók** | ![Combos](docs/screenshots/02-combos.png) | -| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | -| **Egészség** | ![Egészség](docs/screenshots/04-health.png) | -| **Fordító** | ![Translator](docs/screenshots/05-translator.png) | -| **Beállítások** | ![Beállítások](docs/screenshots/06-settings.png) | -| **CLI eszközök** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | -| **Használati naplók** | ![Használat](docs/screenshots/08-usage.png) | -| **Végpontok** | ![Végpontok](docs/screenshots/09-endpoint.png) |
| + --- ### 🤖 Free AI Provider for your favorite coding agents -_Csatlakoztasson bármilyen mesterséges intelligencia-alapú IDE-t vagy CLI-eszközt az OmniRoute-on keresztül – ingyenes API-átjáró a korlátlan kódoláshoz._ - - - - - -OpenClaw
-OpenClaw -

-⭐ 205 000 - - - -NanoBot
-NanoBot -

-⭐ 20,9K - - - -PicoClaw
-PicoClaw -

-⭐ 14,6 KB - - - -ZeroClaw
-ZeroClaw -

-⭐ 9,9 KB - - - -Vaskarom
-Vaskarom -

-⭐ 2,1K - - - - - -OpenCode
-OpenCode -

-⭐ 106K - - - -Codex CLI
-Codex CLI -

-⭐ 60,8K - - - -Claude Code
-Claude Code -

-⭐ 67,3 K - - - -Gemini CLI
-Gemini CLI -

-⭐ 94,7K - - - -Kilókód
-Kilókód -

-⭐ 15,5 KB - - +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ + + + + + + + + + + + + + + +
+ + OpenClaw
+ OpenClaw +

+ ⭐ 205K +
+ + NanoBot
+ NanoBot +

+ ⭐ 20.9K +
+ + PicoClaw
+ PicoClaw +

+ ⭐ 14.6K +
+ + ZeroClaw
+ ZeroClaw +

+ ⭐ 9.9K +
+ + IronClaw
+ IronClaw +

+ ⭐ 2.1K +
+ + OpenCode
+ OpenCode +

+ ⭐ 106K +
+ + Codex CLI
+ Codex CLI +

+ ⭐ 60.8K +
+ + Claude Code
+ Claude Code +

+ ⭐ 67.3K +
+ + Gemini CLI
+ Gemini CLI +

+ ⭐ 94.7K +
+ + Kilo Code
+ Kilo Code +

+ ⭐ 15.5K +
-📡 Minden ügynök a http://localhost:20128/v1 vagy a http://cloud.omniroute.online/v1 segítségével csatlakozik – egy konfiguráció, korlátlan modellek és kvóta--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Ne pazarolja a pénzt, és ne lépje túl a limiteket:** +**Stop wasting money and hitting limits:** -- Az előfizetési kvóta minden hónapban fel nem használt -- A díjkorlátok megakadályozzák a középső kódolást -- Drága API-k (20-50 USD/hó szolgáltatónként) -- Manuális váltás a szolgáltatók között +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**Az OmniRoute ezt megoldja:** +**OmniRoute solves this:** -- ✅**Az előfizetések maximalizálása**- Kövesse nyomon a kvótát, használjon fel minden bitet a visszaállítás előtt -- ✅**Automatikus tartalék**- Előfizetés → API-kulcs → Olcsó → Ingyenes, nulla állásidő -- ✅**Több fiók**- Kör-robin a fiókok között szolgáltatónként -- ✅**Univerzális**- Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, bármilyen CLI eszközzel működik--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Csatlakozzon közösségünkhöz!**[WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) – Kérjen segítséget, ossza meg tippjeit, és maradjon naprakész. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Webhely**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Problémák**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Közösségi csoport](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Hozzájárulás**: Tekintse meg a [CONTRIBUTING.md](CONTRIBUTING.md) oldalt, nyisson PR-t, vagy válasszon egy "jó első számot". -**Original Project**: [9router by decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Egy probléma megnyitásakor futtassa a system-info parancsot, és csatolja a generált fájlt:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Ez létrehoz egy "system-info.txt" fájlt a Node.js verziójával, az OmniRoute verziójával, az operációs rendszer részleteivel, a telepített CLI-eszközökkel (qoder, gemini, claude, codex, antigravitáció, droid stb.), Docker/PM2 állapottal és rendszercsomagokkal – mindennel, amire szükségünk van a probléma gyors reprodukálásához. Csatolja a fájlt közvetlenül a GitHub-problémához.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Minden mesterséges intelligencia-eszközöket használó fejlesztő naponta szembesül ezekkel a problémákkal.**Az OmniRoute úgy készült, hogy ezeket mind megoldja – a költségtúllépésektől a regionális blokkokig, a megszakadt OAuth-folyamatoktól a protokollműveletekig és a vállalati megfigyelhetőségig. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. - -💸 1. "Drága előfizetésért fizetek, de még mindig megzavarnak a korlátok" +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -A fejlesztők havi 20–200 dollárt fizetnek a Claude Pro, Codex Pro vagy GitHub Copilotért. A kvótának még fizetés esetén is van felső határa – 5 óra használat, heti limitek vagy percdíjkorlátok. Mid-coding session, the provider stops responding and the developer loses flow and productivity. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Hogyan oldja meg az OmniRoute:** +**How OmniRoute solves it:** --**Smart 4-Tier Fallback**– Ha az előfizetési kvóta kimerül, automatikusan átirányítja az API-kulcs → Olcsó → Ingyenes, manuális beavatkozás nélkül --**Szolgáltatói korlátozások követése**– A gyorsítótárazott kvóta pillanatképei szerveroldali ütemezés szerint frissülnek (alapértelmezett `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`), a felhasználói felületen elérhető kézi frissítéssel --**Több fiók támogatása**- Több fiók szolgáltatónként automatikus körváltással - ha az egyik elfogy, átvált a következőre --**Egyéni kombinációk**— Testreszabható tartalék láncok 9 kiegyensúlyozási stratégiával (prioritásos, súlyozott, kitöltési sorrendben, körbefutó, P2C, véletlenszerű, legkevésbé használt, költségoptimalizált, szigorúan véletlenszerű) --**Codex üzleti kvóták**— Üzleti/csapat munkaterület-kvóta figyelése közvetlenül az irányítópulton
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard - -🔌 2. "Több szolgáltatót kell használnom, de mindegyik más API-val rendelkezik" + -Az OpenAI egy formátumot használ, a Claude (Anthropic) egy másikat, a Gemini pedig egy másikat. Ha egy fejlesztő különböző szolgáltatók modelljeit szeretné tesztelni, vagy tartalékot szeretne közöttük, akkor újra kell konfigurálnia az SDK-kat, módosítania kell a végpontokat, és kezelnie kell az inkompatibilis formátumokat. Az egyéni szolgáltatók (FriendLI, NIM) nem szabványos modellvégpontokkal rendelkeznek. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Hogyan oldja meg az OmniRoute:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Egységes végpont**- Egyetlen "http://localhost:20128/v1" proxyként szolgál mind a 60+ szolgáltató számára --**Formátumfordítás**- Automatikus és átlátható: OpenAI ↔ Claude ↔ Gemini ↔ Responses API --**Response Sanitization**— Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ --**Szerepek normalizálása**— Átalakítja a "fejlesztő" → "rendszert" a nem OpenAI szolgáltatók számára; `rendszer` → `felhasználó` a GLM/ERNIE-hez --**Think Tag Extraction**– Kivonja a „” blokkokat olyan modellekből, mint a DeepSeek R1 szabványos „reasoning_content” tartalommá --**Strukturált kimenet a Gemini számára**— `json_schema` → `responseMimeType`/`responseSchema` automatikus átalakítás -- A**„stream” alapértelmezése „false”**– Az OpenAI specifikációhoz igazodik, elkerülve a váratlan SSE-t a Python/Rust/Go SDK-kban
+**How OmniRoute solves it:** - -🌐 3. „Az AI-szolgáltatóm blokkolja a régiómat/országomat” +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Az olyan szolgáltatók, mint az OpenAI/Codex, blokkolják a hozzáférést bizonyos földrajzi régiókból. A felhasználók OAuth- és API-kapcsolatok során olyan hibákat kapnak, mint a „nem támogatott_ország_régió_területe”. Ez különösen frusztráló a fejlődő országok fejlesztői számára. + -**Hogyan oldja meg az OmniRoute:** +
+🌐 3. "My AI provider blocks my region/country" --**3-szintű proxykonfiguráció**– 3 szinten konfigurálható proxy: globális (teljes forgalom), szolgáltatónként (csak egy szolgáltató) és kapcsolatonként/kulcsonként --**Színes proxy jelvények**- Vizuális jelzők: 🟢 globális proxy, 🟡 szolgáltató proxy, 🔵 kapcsolat proxy, mindig az IP-t mutatja --**OAuth-tokencsere proxyn keresztül**– Az OAuth-folyamat a proxyn is keresztülmegy, megoldva a `nem támogatott_ország_régió_területét' --**Kapcsolódási tesztek proxyn keresztül**- A csatlakozási tesztek a konfigurált proxyt használják (nincs többé közvetlen kiiktatás) --**SOCKS5 támogatás**— Teljes SOCKS5 proxy támogatás a kimenő útválasztáshoz --**TLS-ujjlenyomat-hamisítás**– Böngészőszerű TLS-ujjlenyomat a "wreq-js"-n keresztül a botészlelés megkerüléséhez --**🔏 CLI Ujjlenyomat Matching**– A fejlécek és törzsmezők átrendezése, hogy megfeleljenek a natív CLI bináris aláírásoknak, drasztikusan csökkentve a fiók megjelölésének kockázatát. A proxy IP-címe megmarad – egyszerre kapja meg a lopakodó**és**IP-maszkolást
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. - -🆓 4. "MI-t akarok használni kódoláshoz, de nincs pénzem" +**How OmniRoute solves it:** -Nem mindenki fizethet havi 20–200 dollárt az AI-előfizetésekért. A feltörekvő országok diákjainak, fejlesztőinek, amatőröknek és szabadúszóknak nulla költséggel kell hozzáférniük a minőségi modellekhez. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Hogyan oldja meg az OmniRoute:** + --**Beépített ingyenes szolgáltatók**- Natív támogatás 100%-ban ingyenes szolgáltatókhoz: Qoder (5 korlátlan modell OAuth-on keresztül: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlim-limitus3 modell) qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID ingyen), Gemini CLI (180 000 token/hónap ingyenes) --**Ollama Cloud**– Felhőben tárolt Ollama-modellek az `api.ollama.com-on, ingyenes "Light usage" szinttel; használja az `ollamacloud/` előtagot --**Csak ingyenes kombók**— Lánc `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = 0 USD/hó nulla állásidővel --**NVIDIA NIM ingyenes hozzáférés**– ~40 RPM fejlesztői örökké ingyenes hozzáférés több mint 70 modellhez a build.nvidia.com oldalon (áttérés a kreditekről a tiszta sebességkorlátokra) --**Költségoptimalizált stratégia**— Útválasztási stratégia, amely automatikusan a legolcsóbb elérhető szolgáltatót választja +
+🆓 4. "I want to use AI for coding but I have no money" - -🔒 5. "Meg kell védenem a mesterséges intelligencia átjárómat a jogosulatlan hozzáféréstől" +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -Ha AI átjárót teszünk ki a hálózatnak (LAN, VPS, Docker), a cím birtokában bárki felhasználhatja a fejlesztő tokenjeit/kvótáját. Védelem nélkül az API-k sebezhetőek a visszaélésekkel, azonnali befecskendezéssel és visszaélésekkel szemben. +**How OmniRoute solves it:** -**Hogyan oldja meg az OmniRoute:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**API-kulcskezelés**– Generálás, rotáció és hatókör szolgáltatónként egy dedikált "/dashboard/api-manager" oldallal --**Modellszintű engedélyek**– Az API-kulcsok korlátozása adott modellekre ("openai/*", helyettesítő karakteres minták), az Összes engedélyezése/Korlátozása kapcsolóval --**API Endpoint Protection**– Kulcs szükséges a `/v1/models' számára, és bizonyos szolgáltatók letiltása a listából --**Auth Guard + CSRF védelem**- Minden műszerfali útvonal "withAuth" köztes szoftverrel + CSRF tokenekkel védett --**Rate Limiter**— IP-nkénti sebességkorlátozás konfigurálható ablakokkal --**IP-szűrés**— Engedélyezési lista/blokkolólista a hozzáférés-vezérléshez --**Prompt Injection Guard**– fertőtlenítés a rosszindulatú felszólítási minták ellen --**AES-256-GCM titkosítás**- A hitelesítő adatok nyugalmi állapotban titkosítva
+ - -🛑 6. "A szolgáltatóm leállt, és elvesztettem a kódolási folyamatomat" +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -Az AI-szolgáltatók instabillá válhatnak, 5xx-es hibákat adnak vissza, vagy elérhetik az ideiglenes sebességkorlátokat. Ha egy fejlesztő egyetlen szolgáltatótól függ, akkor megszakad. Megszakítók nélkül az ismételt újrapróbálkozások összeomolhatják az alkalmazást. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Hogyan oldja meg az OmniRoute:** +**How OmniRoute solves it:** --**Megszakító típusonként**- Automatikus nyitás/zárás konfigurálható küszöbértékekkel és lehűtéssel (zárt/nyitott/félig nyitott), modellenkénti hatókör a lépcsőzetes blokkok elkerülése érdekében --**Exponenciális visszalépés**— Progresszív újrapróbálkozási késések --**Mennydörgés elleni csorda**- Mutex + szemafor védelem az egyidejű újrapróbálkozási viharok ellen --**Kombinált tartalék láncok**– Ha az elsődleges szolgáltató meghibásodik, automatikusan, beavatkozás nélkül átesik a láncon --**Combo Circuit Breaker**– Automatikusan letiltja a hibás szolgáltatókat a kombinált láncon belül --**Egészségügyi irányítópult**— Üzemidő-figyelés, áramkör-megszakító állapotok, zárolások, gyorsítótár-statisztika, p50/p95/p99 késleltetés
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest - -🔧 7. "Az egyes AI-eszközök konfigurálása fárasztó és ismétlődő." + -A fejlesztők Cursort, Claude Code-ot, Codex CLI-t, OpenClaw-ot, Gemini CLI-t, Kilo Code-ot használnak... Minden eszköznek más konfigurációra van szüksége (API végpont, kulcs, modell). Az újrakonfigurálás szolgáltató- vagy modellváltáskor időpocsékolás. +
+🛑 6. "My provider went down and I lost my coding flow" -**Hogyan oldja meg az OmniRoute:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**CLI Tools Dashboard**- Dedikált oldal egykattintásos beállítással a Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline számára --**GitHub Copilot Config Generator**– A "chatLanguageModels.json" fájlt generálja a VS-kódhoz tömeges modellválasztással --**Bevezető varázsló**– Irányított 4 lépéses beállítás első felhasználók számára --**Egy végpont, minden modell**- Konfigurálja egyszer a `http://localhost:20128/v1' címet, elérje a 60+ szolgáltatót
+**How OmniRoute solves it:** - -🔑 8. „A több szolgáltatótól származó OAuth-tokenek kezelése pokol” +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot – mindegyik az OAuth 2.0-t használja lejáró tokenekkel. A fejlesztőknek folyamatosan újra kell hitelesíteniük, kezelniük kell a „kliens_titka hiányzik”, az „átirányítási_uri_mismatch” és a távoli szerverek hibáival. Az OAuth a LAN/VPS-en különösen problémás. + -**Hogyan oldja meg az OmniRoute:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Automatikus tokenfrissítés**- Az OAuth-tokenek a háttérben frissülnek a lejárat előtt --**OAuth 2.0 (PKCE) beépített**- Automatikus áramlás Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder számára --**Multi-Account OAuth**- Több fiók szolgáltatónként a JWT/ID token kivonattal --**OAuth LAN/Távoli javítás**– Privát IP-észlelés az „átirányítási_uri”-hoz + kézi URL mód távoli szerverekhez --**OAuth Nginx mögött**– A "window.location.origin" fájlt használja a fordított proxy kompatibilitás érdekében --**Távoli OAuth útmutató**– Lépésről lépésre útmutató a Google Cloud hitelesítő adataihoz VPS/Docker rendszeren
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. - -📊 9. "Nem tudom, mennyit költök és hova" +**How OmniRoute solves it:** -A fejlesztők több fizetős szolgáltatót használnak, de nincs egységes nézetük a kiadásokról. Minden szolgáltató saját számlázási irányítópulttal rendelkezik, de nincs összevont nézet. A váratlan költségek felhalmozódhatnak. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Hogyan oldja meg az OmniRoute:** + --**Költségelemzési irányítópult**– Tokenenkénti költségkövetés és költségkeret-kezelés szolgáltatónként --**Költségkeret-korlátok rétegenként**- Költési felső határ szintenként, amely automatikus visszalépést vált ki --**Modellenkénti árképzés**- Konfigurálható árak modellenként --**Használati statisztika API-kulcsonként**— A kérések száma és az utoljára használt időbélyeg kulcsonként --**Analytics Dashboard**— Statisztikai kártyák, modellhasználati diagram, szolgáltatói táblázat sikerarányokkal és késleltetéssel +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" - -🐛 10. "Nem tudom diagnosztizálni a hibákat és problémákat az AI-hívásoknál" +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Ha egy hívás meghiúsul, a fejlesztő nem tudja, hogy sebességkorlátozás, lejárt token, rossz formátum vagy szolgáltatói hiba volt-e. Töredezett naplók a különböző terminálokon. Megfigyelhetőség nélkül a hibakeresés próba és hiba. +**How OmniRoute solves it:** -**Hogyan oldja meg az OmniRoute:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Egységes naplók irányítópultja**- 4 lap: Kérelemnaplók, Proxynaplók, Auditnaplók, Konzol --**Konzolnapló-nézegető**— Valós idejű terminál stílusú megjelenítő színkódolt szintekkel, automatikus görgetés, keresés, szűrés --**SQLite proxynaplók**– Állandó naplók, amelyek túlélik a szerver újraindítását --**Translator Playground**– 4 hibakeresési mód: Playground (formátum fordítás), Chat Tester (oda-vissza út), Tesztpad (kötegelt), Élő monitor (valós idejű) --**Request Telemetria**– p50/p95/p99 késleltetés + X-Request-Id nyomkövetés --**File-Based Logging with Rotation**— App logs rotate by size, retention days, and archive count; a hívásnapló műtermékei a megőrzési napok és a fájlok száma szerint váltakoznak --**Rendszerinformációs jelentés**— Az `npm run system-info` létrehozza a `system-info.txt' fájlt a teljes környezettel (Node verzió, OmniRoute verzió, OS, CLI eszközök, Docker/PM2 állapot). Csatolja, amikor az azonnali osztályozással kapcsolatos problémákat jelent be.
+ - -🏗️ 11. „Az átjáró telepítése és karbantartása bonyolult” +
+📊 9. "I don't know how much I'm spending or where" -Az AI-proxy telepítése, konfigurálása és karbantartása különböző környezetekben (helyi, VPS, Docker, felhő) munkaigényes. Az olyan problémák, mint a keménykódolt elérési utak, az „EACCES” a könyvtárakon, a portkonfliktusok és a többplatformos buildek súrlódást okoznak. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Hogyan oldja meg az OmniRoute:** +**How OmniRoute solves it:** --**npm globális telepítés**— `npm install -g omniroute && omniroute` — kész --**Docker Multi-Platform**– AMD64 + ARM64 natív (Apple Silicon, AWS Graviton, Raspberry Pi) --**Docker Compose Profiles**– "alap" (nincs CLI-eszközök) és "cli" (Claude Code, Codex, OpenClaw) --**Electron Desktop App**– Natív alkalmazás Windows/macOS/Linux rendszerhez rendszertálcával, automatikus indítással, offline móddal --**Split-Port Mode**– API és irányítópult külön portokon haladó forgatókönyvekhez (fordított proxy, konténerhálózat) --**Cloud Sync**– Szinkronizálás konfigurálása az eszközök között a Cloudflare Workers segítségével --**DB biztonsági mentések**– Az összes beállítás automatikus biztonsági mentése, visszaállítása, exportálása és importálása, a `DISABLE_SQLITE_AUTO_BACKUP` funkcióval a külsőleg kezelt biztonsági mentésekhez
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency - -🌍 12. "A felület csak angol nyelvű, és a csapatom nem beszél angolul" + -A nem angol nyelvű országok csapatai, különösen Latin-Amerikában, Ázsiában és Európában, csak angol nyelvű felületekkel küszködnek. A nyelvi akadályok csökkentik az átvételt és növelik a konfigurációs hibákat. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Hogyan oldja meg az OmniRoute:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Irányítópult i18n – 30 nyelv**– Mind az 500+ billentyű lefordítva, beleértve arab, bolgár, dán, német, spanyol, finn, francia, héber, hindi, magyar, indonéz, olasz, japán, koreai, maláj, holland, norvég, lengyel, portugál (PT/BR), román, thai, orosz, szlovák, svéd, filippínó, angol, thai, orosz, kínai, filippínó --**RTL támogatás**– Jobbról balra haladó arab és héber nyelv támogatása --**Többnyelvű README-k**— 30 teljes dokumentáció fordítás --**Nyelvválasztó**— Globe ikon a fejlécben a valós idejű váltáshoz
+**How OmniRoute solves it:** - -🔄 13. "Többre van szükségem, mint csevegésre – beágyazásra, képekre, hanganyagra van szükségem" +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -Az AI nem csak a csevegés befejezése. A fejlesztőknek képeket kell generálniuk, hangot kell átírniuk, beágyazást kell létrehozniuk a RAG számára, át kell sorolniuk a dokumentumokat, és moderálniuk kell a tartalmat. Minden API más végponttal és formátummal rendelkezik. + -**Hogyan oldja meg az OmniRoute:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Beágyazások**– `/v1/beágyazások' 6 szolgáltatóval és 9+ modellel --**Képgenerálás**– `/v1/images/generations' 10 szolgáltatóval és 20+ modellel (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Text-to-Video**— `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) és SD WebUI --**Text-to-Music**— `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) --**Audio átírás**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Text-to-Speech**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + meglévő szolgáltatók --**Moderációk**— `/v1/moderations` — Tartalombiztonsági ellenőrzések --**Reranging**— `/v1/rerank` — Dokumentumreleváns átsorolás --**Responses API**- Teljes `/v1/responses` támogatás a Codexhez
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. - -🧪 14. "Nincs módom tesztelni és összehasonlítani a minőséget a különböző modellek között" +**How OmniRoute solves it:** -A fejlesztők szeretnék tudni, hogy melyik modell a legjobb az ő használati esetükben – kód, fordítás, érvelés –, de a manuális összehasonlítás lassú. Nincsenek integrált eval eszközök. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**Hogyan oldja meg az OmniRoute:** + --**LLM-értékelések**— Arany készlet tesztelése 10 előre betöltött esettel, beleértve az üdvözlést, a matematikát, a földrajzot, a kódgenerálást, a JSON-megfelelőséget, a fordítást, a leértékelést, a biztonsági megtagadást --**4 egyezési stratégia**– "pontos", "tartalmazza", "regex", "egyéni" (JS függvény) --**Translator Playground Test Bench**- Kötegelt tesztelés több bemenettel és várható kimenettel, szolgáltatók közötti összehasonlítás --**Csevegés tesztelő**- Teljes körút vizuális válaszmegjelenítéssel --**Élő monitor**– Valós idejű adatfolyam a proxyn keresztül folyó összes kérésről +
+🌍 12. "The interface is English-only and my team doesn't speak English" - -📈 15. "A teljesítmény elvesztése nélkül kell skáláznom" +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -A kérelmek mennyiségének növekedésével ugyanazok a kérdések gyorsítótárazás nélkül duplikált költségeket generálnak. Idempotencia nélkül a duplikált hulladékfeldolgozási kérelmek. A szolgáltatónkénti díjkorlátokat be kell tartani. +**How OmniRoute solves it:** -**Hogyan oldja meg az OmniRoute:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Szemantikus gyorsítótár**– A kétszintű gyorsítótár (aláírás + szemantikai) csökkenti a költségeket és a késleltetést --**Idempotency kérése**– 5 másodperces deduplikációs ablak azonos kérések esetén --**Drátakorlát észlelése**– Szolgáltatónkénti RPM, minimális rés és maximális egyidejű követés --**Szerkeszthető sebességkorlátok**- Konfigurálható alapértékek a Beállítások → Kitartással ellenálló képesség menüpontban --**API Key Validation Cache**– 3-szintű gyorsítótár az éles teljesítményhez --**Egészségügyi irányítópult telemetriával**— p50/p95/p99 késleltetés, gyorsítótár statisztika, üzemidő
+ - -🤖 16. "Globálisan szeretném irányítani a modell viselkedését" +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Azok a fejlesztők, akik minden választ egy adott nyelven, egy adott hangnemben szeretnének, vagy korlátozni szeretnék az érvelési tokeneket. Ennek konfigurálása minden eszközben/kérelemben nem praktikus. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Hogyan oldja meg az OmniRoute:** +**How OmniRoute solves it:** --**Rendszerprompt Injection**— Globális prompt minden kérelemre vonatkozik --**A költségkeret átgondolásának ellenőrzése**– Indoklási token-kiosztás ellenőrzése kérésenként (áthaladó, automatikus, egyéni, adaptív) --**9 Routing Strategies**– Globális stratégiák, amelyek meghatározzák a kérések elosztását --**Wildcard Router**– a „szolgáltató/*” minták dinamikusan továbbítanak bármely szolgáltatóhoz --**Kombinációs engedélyezés/letiltás váltás**- A kombók váltása közvetlenül az irányítópultról --**Provider Toggle**— Egy szolgáltató összes kapcsolatának engedélyezése/letiltása egyetlen kattintással --**Letiltott szolgáltatók**- Adott szolgáltatók kizárása a `/v1/models' listáról
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex - -🧰 17. "Szükségem van az MCP-eszközökre, mint első osztályú termékképességekre" + -Sok mesterséges intelligencia-átjáró csak rejtett megvalósítási részletként teszi közzé az MCP-t. A csapatoknak látható, kezelhető műveleti rétegre van szükségük. +
+🧪 14. "I have no way to test and compare quality across models" -**Hogyan oldja meg az OmniRoute:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- Az MCP megjelenik az irányítópult navigációs és végponti protokoll lapján -- Dedikált MCP-kezelési oldal folyamatokkal, eszközökkel, hatókörökkel és audittal -- Beépített gyorsindítás az "omniroute --mcp" és a kliens beépítéséhez
+**How OmniRoute solves it:** - -🧠 18. "A2A hangszerelésre van szükségem szinkronizálással + adatfolyam feladatútvonalak" +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Az ügynöki munkafolyamatokhoz közvetlen válaszokra és hosszú távú, streamelt végrehajtásra van szükség életciklus-vezérléssel. + -**Hogyan oldja meg az OmniRoute:** +
+📈 15. "I need to scale without losing performance" -- A2A JSON-RPC végpont ("POST /a2a") "message/send" és "message/stream" paraméterekkel -- SSE streaming terminál állapot terjesztéssel -- Feladatéletciklus API-k a "tasks/get" és a "tasks/cancel" számára
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. - -🛰️ 19. "Valódi MCP-folyamat-állapotra van szükségem, nem kitalált állapotra" +**How OmniRoute solves it:** -Az operatív csapatoknak tudniuk kell, hogy az MCP valóban életben van-e, nem csak azt, hogy egy API elérhető-e. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**Hogyan oldja meg az OmniRoute:** + -- Futásidejű szívverés fájl PID-vel, időbélyegekkel, szállítással, szerszámszámmal és hatókör móddal -- MCP állapot API, amely kombinálja a szívverést + a legutóbbi tevékenységet -- UI állapotkártyák a folyamat/üzemidő/szívverés frissességéhez +
+🤖 16. "I want to control model behavior globally" - -📋 20. "Kivizsgálható MCP-eszköz-végrehajtásra van szükségem" +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Amikor az eszközök módosítják a konfigurációt vagy működési műveleteket indítanak el, a csapatoknak kriminalisztikai nyomon követhetőségre van szükségük. +**How OmniRoute solves it:** -**Hogyan oldja meg az OmniRoute:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- SQLite-alapú auditnaplózás MCP-eszközhívásokhoz -- Szűrések eszköz, siker/kudarc, API-kulcs és oldalszámozás szerint -- Irányítópult audit táblázat + statisztikai végpontok az automatizáláshoz
+ - -🔐 21. "Integrációnként hatókörű MCP-engedélyekre van szükségem" +
+🧰 17. "I need MCP tools as first-class product capabilities" -A különböző ügyfeleknek a legkevesebb jogosultsággal kell rendelkezniük az eszközkategóriákhoz. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Hogyan oldja meg az OmniRoute:** +**How OmniRoute solves it:** -- 10 szemcsés MCP hatókör az ellenőrzött szerszámhozzáféréshez -- Hatályérvényesítés és láthatóság az MCP-kezelő felületen -- Biztonságos alaphelyzet az üzemi szerszámokhoz
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding - -⚙️ 22. "Üzemeltetési vezérlőkre van szükségem átcsoportosítás nélkül" + -A csapatoknak gyors futásidejű változtatásokra van szükségük incidensek vagy költségesemények során. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**Hogyan oldja meg az OmniRoute:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- A kombinált aktiválás váltása közvetlenül az MCP műszerfaláról -- Rugalmassági profilok alkalmazása előre meghatározott házirend-csomagokból -- Állítsa vissza a megszakító állapotát ugyanarról a kezelőpanelről
+**How OmniRoute solves it:** - -🔄 23. "Szükségem van az élő A2A feladat életciklusának láthatóságára és törlésére" +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Az életciklus láthatósága nélkül a feladat-incidensek nehezen osztályozhatók. + -**Hogyan oldja meg az OmniRoute:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Feladatok listázása/szűrés állapot/készség szerint oldalszámozással -- A feladatok metaadatainak, eseményeinek és műtermékeinek részletezése -- Feladat törlési végpont és felhasználói felület művelet megerősítéssel
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. - -🌊 24. "Aktív adatfolyam-metrikákra van szükségem A2A terheléshez" +**How OmniRoute solves it:** -A streamelési munkafolyamatok működési betekintést igényelnek a párhuzamosság és az élő kapcsolatok terén. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**Hogyan oldja meg az OmniRoute:** + -- Az A2A állapotba integrált aktív folyamszámlálók -- Utolsó feladat időbélyegzője és állapotonkénti száma -- A2A műszerfalkártyák a valós idejű műveletek figyeléséhez +
+📋 20. "I need auditable MCP tool execution" - -🪪 25. "Szabványos ügynökfelderítésre van szükségem az ügyfelek számára" +When tools mutate config or trigger ops actions, teams need forensic traceability. -A külső klienseknek és hangszerelőknek géppel olvasható metaadatokra van szükségük a bevezetéshez. +**How OmniRoute solves it:** -**Hogyan oldja meg az OmniRoute:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- Az ügynökkártya elérhető a `/.well-known/agent.json' címen -- A menedzsment felületen látható képességek és készségek -- Az A2A állapot API felfedezési metaadatokat tartalmaz az automatizáláshoz
+ - -🧭 26. "Protokoll felfedezhetőségre van szükségem a termék felhasználói élményében" +
+🔐 21. "I need scoped MCP permissions per integration" -Ha a felhasználók nem fedezik fel a protokollfelületeket, az elfogadás és a támogatás minősége csökken. +Different clients should have least-privilege access to tool categories. -**Hogyan oldja meg az OmniRoute:** +**How OmniRoute solves it:** -- Összevont**Végpontok**oldal a proxy, MCP, A2A és API végpontok lapjaival -- Inline szolgáltatás állapotát váltja (Online/Offline) MCP és A2A esetén -- Hivatkozások az áttekintésből a dedikált kezelőlapokhoz
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling - -🧪 27. "Végponttól végpontig terjedő protokoll-érvényesítésre van szükségem valódi ügyfelekkel" + -A próbatesztek nem elegendőek a protokoll-kompatibilitás ellenőrzéséhez a kiadás előtt. +
+⚙️ 22. "I need operational controls without redeploying" -**Hogyan oldja meg az OmniRoute:** +Teams need quick runtime changes during incidents or cost events. -- E2E csomag, amely elindítja az alkalmazást, és valódi MCP SDK kliens szállítást használ -- Az A2A kliens teszteli az áramlások felfedezését, küldését, streamingjét, lekérését és megszakítását -- Az állítások keresztellenőrzése az MCP audit és az A2A feladatok API-jával szemben
+**How OmniRoute solves it:** - -📡 28. "Egységes megfigyelhetőségre van szükségem az összes felületen" +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -A megfigyelhetőség protokoll szerinti felosztása vakfoltokat és hosszabb MTTR-t hoz létre. + -**Hogyan oldja meg az OmniRoute:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Egységes irányítópultok/naplók/analytics egy termékben -- Egészség + audit + kérés telemetria OpenAI, MCP és A2A rétegeken keresztül -- Működési API-k az állapothoz és az automatizáláshoz
+Without lifecycle visibility, task incidents become hard to triage. - -💼 29. "Egy futási időre van szükségem a proxy + eszközök + ügynök hangszereléshez" +**How OmniRoute solves it:** -Számos külön szolgáltatás futtatása növeli a működési költségeket és a hibamódokat. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**Hogyan oldja meg az OmniRoute:** + -- OpenAI-kompatibilis proxy, MCP szerver és A2A szerver egy veremben -- Megosztott hitelesítés, rugalmasság, adattárolás és megfigyelhetőség -- Konzisztens politikai modell az összes interakciós felületen +
+🌊 24. "I need active stream metrics for A2A load" - -🚀 30. "Ügynöki munkafolyamatokat ragasztókód szétterülése nélkül kell szállítanom" +Streaming workflows require operational insight into concurrency and live connections. -A csapatok veszítenek sebességükből, amikor több ad-hoc szolgáltatást és szkriptet illesztenek össze. +**How OmniRoute solves it:** -**Hogyan oldja meg az OmniRoute:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Egységes végpont stratégia az ügyfelek és ügynökök számára -- Beépített protokollkezelő felhasználói felületek és füstellenőrzési útvonalak -- Gyártásra kész alapok (biztonság, naplózás, rugalmasság, biztonsági mentés)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**A játékkönyv: Maximalizálja a fizetett előfizetést + olcsó biztonsági mentés**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -609,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Playbook B: Zéró költségű kódolási verem**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Playbook C: 24/7 mindig bekapcsolt tartalék lánc**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -632,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**D játékkönyv: Az ügynök MCP + A2A-val működik**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Állítsa be az AI-kódolást percek alatt**0 USD/hó**áron. Csatlakoztassa ezeket az ingyenes fiókokat, és használja a beépített**Free Stack**kombinációt. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| lépés | Akció | Szolgáltatók feloldva | -| ---- | --------------------------------------------------- | ------------------------------------------------------------------ | -| 1 | Csatlakozás**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**korlátlan**| -| 2 | Csatlakozás**Qoder**(Google OAuth) | kimi-k2-gondolkodás, qwen3-coder-plus, deepseek-r1... —**korlátlan**| -| 3 | Csatlakoztassa a**Qwen**(eszközkód) | qwen3-coder-plus, qwen3-coder-flash... —**korlátlan**| -| 4 | Csatlakozás**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro –**180K/hó ingyenes**| -| 5 | `/dashboard/combos` →**Ingyenes köteg ($0)**sablon | Körbe-körbe minden ingyenes szolgáltató automatikusan | +| Step | Action | Providers Unlocked | +| ---- | -------------------------------------------------- | ------------------------------------------------------------------ | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Mutasson bármely IDE/CLI-t a következőre:**`http://localhost:20128/v1` · API-kulcs: `any-string` · Kész. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Opcionális extra lefedettség (szintén ingyenes):**Groq API kulcs (30 RPM ingyenes), NVIDIA NIM (40 RPM ingyenes, 70+ modell), Cerebras (1 millió tok/nap), LongCat API kulcs (50 millió token/nap!), Cloudflare Workers AI (10 000 neuron/nap, 50+ modell).## Gyors kezdés +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Gyors kezdés ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **pnpm felhasználók:**Telepítés után futtassa a `pnpm approve-builds -g` parancsot, hogy engedélyezze a `better-sqlite3` és a `@swc/core` által igényelt natív build szkripteket: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > > ```bash > pnpm install -g omniroute -> pnpm approve-builds -g # Válassza ki az összes csomagot → jóváhagyja +> pnpm approve-builds -g # Select all packages → approve > omniroute > ``` -Az irányítópult a „http://localhost:20128” címen nyílik meg, az API alap URL-címe pedig „http://localhost:20128/v1”. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Parancs | Leírás | -| ----------------------- | ----------------------------------------------------------------------- | -| `omniroute` | Szerver indítása (`PORT=20128`, API és irányítópult ugyanazon a porton) | -| `omniroute --port 3000` | A kanonikus/API port beállítása 3000 | -| `omniroute --mcp` | MCP-kiszolgáló indítása (stdio szállítás) | -| `omniroute --no-open` | Ne nyissa meg automatikusan a böngészőt | -| `omniroute --help` | Segítség megjelenítése | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Opcionális osztott portos mód:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -A legtöbb telepítéshez csak a következőkre van szüksége: +For most deployments, you only need: -| Változó | Alapértelmezett | Cél | -| ------------------------- | ------------------------------ | -------------------------------------------------------------- -------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | "600000" | Megosztott alapvonal az upstream lekéréshez, a rejtett Undici-időtúllépésekhez, a TLS-ujjlenyomat-kérésekhez és az API-híd kérés/proxy időtúllépéséhez | -| `STREAM_IDLE_TIMEOUT_MS` | örökli a `REQUEST_TIMEOUT_MS' | Maximális hézag a streaming darabok között, mielőtt az OmniRoute megszakítja az SSE adatfolyamot | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -A visszamenőleges kompatibilitás megmarad: a meglévő „FETCH_TIMEOUT_MS”, „API_BRIDGE_PROXY_TIMEOUT_MS” és más rétegenkénti időtúllépési változók továbbra is működnek, és felülírják a megosztott alapvonalat. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Speciális felülírások állnak rendelkezésre, ha finomabb vezérlésre van szüksége:| Változó | Alapértelmezett | Cél | -| ----------------------------------------- | ------------------------------------------- | --------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | örökli a `REQUEST_TIMEOUT_MS' | A fő lekérés megszakítási jele által használt teljes felfelé irányuló kérés időtúllépése | -| `FETCH_HEADERS_TIMEOUT_MS` | örökli a `FETCH_TIMEOUT_MS` | Undici időkorlát az upstream válaszfejlécek fogadására | -| `FETCH_BODY_TIMEOUT_MS` | örökli a `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | -| `FETCH_CONNECT_TIMEOUT_MS` | "30000" | Undici TCP csatlakozási időtúllépés | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | "4000" | Undici tétlen életben tartási aljzat időtúllépése | -| `TLS_CLIENT_TIMEOUT_MS` | örökli a `FETCH_TIMEOUT_MS` | Időtúllépés a `wreq-js` | segítségével küldött TLS-ujjlenyomat-kéréseknél -| `API_BRIDGE_PROXY_TIMEOUT_MS` | örökli a „REQUEST_TIMEOUT_MS” vagy „30000” | Időtúllépés a „/v1” proxy API-portról az irányítópult-portra való továbbítására | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | "max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)" | Bejövő kérés időtúllépése az API-hídszerveren | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | "60000" | Bejövő fejléc időtúllépése az API-hídszerveren | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | "5000" | Életben tartás időtúllépés az API-hídszerveren | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | "0" | Socket inaktivitási időtúllépése az API-hídszerveren ("0" letiltja) | +Advanced overrides are available if you need finer control: -Ha az OmniRoute alkalmazást az Nginx, Caddy, Cloudflare vagy más fordított proxy mögött futtatja, győződjön meg arról, hogy a proxy -az időtúllépések is magasabbak, mint az OmniRoute adatfolyam/lekérési időkorlátok.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Nyissa meg az Irányítópult → „Szolgáltatók” menüpontot, és csatlakoztasson legalább egy szolgáltatót (OAuth- vagy API-kulcs). -2. Nyissa meg a Dashboard → `Végpontok` menüpontot, és hozzon létre egy API-kulcsot. -3. (Opcionális) Nyissa meg az Irányítópult → Kombók menüpontot, és állítsa be a tartalék láncot.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode és OpenAI-kompatibilis SDK-kkal működik.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (szerszámvezérelt műveletekhez):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -Ezután csatlakoztassa MCP-kliensét `stdio'-n keresztül, és tesztelje az olyan eszközöket, mint: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (ügynök-ügynök munkafolyamatokhoz):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -761,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Ez a csomag érvényesíti a valódi MCP- és A2A-kliensfolyamokat egy futó alkalmazással szemben.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -769,14 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` - +
+Void Linux (`xbps-src` template) -Érvénytelen Linux (`xbps-src` sablon) - -Void Linux felhasználók számára natív csomagot készíthet az `xbps-src` használatával. Mentse ezt a blokkot `srcpkgs/omniroute/template' néven:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -788,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -796,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -872,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -883,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -Az OmniRoute nyilvános Docker-képként érhető el a [Docker Hubon](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Gyors futás:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -893,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**Környezetfájllal:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**A Docker Compose használata:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -A Docker-telepítések irányítópult-támogatása mostantól magában foglal egy egykattintásos**Cloudflare Quick Tunnel**-t az "Irányítópult → Végpontok" oldalon. Az első engedélyezés csak szükség esetén tölti le a „cloudflared” funkciót, ideiglenes alagutat indít a jelenlegi „/v1” végponthoz, és megjeleníti a generált „https://\*.trycloudflare.com/v1” URL-t közvetlenül a normál nyilvános URL alatt. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Megjegyzések: +Notes: -- A Quick Tunnel URL-ek ideiglenesek, és minden újraindítás után megváltoznak. -- A gyors alagutak nem állnak vissza automatikusan az OmniRoute vagy a tároló újraindítása után. Ha szükséges, engedélyezze őket újra az irányítópulton. -- A felügyelt telepítés jelenleg támogatja a Linuxot, a macOS-t és a Windowst „x64” / „arm64” rendszeren. -- A felügyelt gyorsalagutak alapértelmezés szerint a HTTP/2 átvitelt használják, hogy elkerüljék a zajos QUIC UDP puffer figyelmeztetéseket a korlátozott tárolókörnyezetekben. Állítsa be a "CLOUDFLARED_PROTOCOL=quic" vagy az "auto" értéket, ha más átvitelt szeretne. -- A Docker képek a rendszer CA-gyökereit csomagolják, és átadják a felügyelt "cloudflared"-nek, amely elkerüli a TLS-megbízhatósági hibákat, amikor az alagút a tárolón belül bootstradik. -- Az SQLite WAL módban fut. Engedélyezni kell a `docker stop' befejezését, hogy az OmniRoute vissza tudja irányítani a legutóbbi változtatásokat a `storage.sqlite' fájlba. -- A kötegelt Compose-fájlok már beállítottak egy 40 másodperces türelmi időt. Ha közvetlenül futtatja a képet, tartsa be a "--stop-timeout 40" értéket (vagy hasonlót), hogy a kézi leállítások ne szakítsák meg a leállítási tisztítást. -- Állítsa be a `CLOUDFLARED_BIN=/absolute/path/to/cloudflared' értéket, ha azt szeretné, hogy az OmniRoute egy létező binárist használjon a letöltés helyett. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**A Docker Compose with Caddy (HTTPS Auto-TLS) használata:** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -Az OmniRoute biztonságosan elérhető a Caddy automatikus SSL-kiépítésével. Győződjön meg arról, hogy a domain DNS-rekordja a szerver IP-címére mutat.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` +| Image | Tag | Size | Description | +| ------------------------ | -------- | ------ | --------------------- | +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | -| Kép | Címke | Méret | Leírás | -| ------------------------- | -------- | ------ | ---------------------- | -| "diegosouzapw/omniroute" | "legújabb" | ~250 MB | Legújabb stabil kiadás | -| "diegosouzapw/omniroute" | "1.0.3" | ~250 MB | Jelenlegi verzió |--- +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**ÚJ!**Az OmniRoute már elérhető**natív asztali alkalmazásként**Windows, macOS és Linux rendszeren. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. Az elektronalapú alkalmazás a következőket tartalmazza: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Natív ablak**- Dedikált alkalmazásablak rendszertálca-integrációval -- 🔄**Automatikus indítás**- Indítsa el az OmniRoute alkalmazást a rendszerbe való bejelentkezéskor -- 🔔**Natív értesítések**- Értesítést kaphat a kvóta kimerüléséről vagy a szolgáltatói problémákról -- ⚡**Egykattintásos telepítés**- NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Offline mód**- Teljesen offline módban működik a mellékelt szerverrel### Gyors kezdés +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Gyors kezdés ```bash # Development mode @@ -982,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -Ha minimalizálja, az OmniRoute a tálcán él, gyors műveletekkel: +When minimized, OmniRoute lives in your system tray with quick actions: -- Nyissa meg a műszerfalat -- Szerver port módosítása -- Lépjen ki az alkalmazásból +- Open dashboard +- Change server port +- Quit application -📖 Teljes dokumentáció: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Tier | Szolgáltató | Költség | Kvóta visszaállítása | Legjobb a | -| ----------------- | ----------------------------- | ---------------------------------- | ---------------------- | --------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -| **💳 ELŐFIZETÉS** | Claude Code (Pro) | 20 USD/hó | 5 óra + heti | Már előfizetett | -| | Codex (Plus/Pro) | 20-200 USD/hó | 5 óra + heti | OpenAI felhasználók | -| | Gemini CLI | **INGYENES** | 180 000/hó + 1 000/nap | Mindenki! | -| | GitHub másodpilóta | 10-19 USD/hó | Havi | GitHub felhasználók | -| **🔑 API KEY** | NVIDIA NIM | **INGYENES**(végre fejlesztő) | ~40 RPM | 70+ nyitott modell | -| | Cerebrák | **INGYENES**(1 millió tok/nap) | 60K TPM / 30 RPM | A világ leggyorsabb | -| | Groq | **INGYENES**(30 RPM) | 14,4K RPD | Ultragyors Llama/Gemma | -| | DeepSeek V3.2 | 0,27 USD/1,10 USD/1 millió | Nincs | Legjobb ár/minőség érvelés | -| | xAI Grok-4 Fast | **0,20 USD/0,50 USD/1 millió**🆕 | Nincs | Leggyorsabb + szerszámhívás, ultralow | -| | xAI Grok-4 (standard) | 0,20 USD/1,50 USD/1M 🆕 | Nincs | Oktatás zászlóshajója az xAI-tól | -| | Mistral | Ingyenes próbaverzió + fizetett | Ár korlátozott | Európai AI | -| | OpenRouter | Felhasználásonkénti fizetés | Nincs | 100+ modell aggr. | -| **💰 OLCSÓ** | GLM-5 (a Z.AI-n keresztül) 🆕 | 0,5 USD/1M | Naponta 10:00 | 128K teljesítmény, legújabb zászlóshajó | -| | GLM-4.7 | 0,6 USD/1M | Naponta 10:00 | Költségvetési biztonsági mentés | -| | MiniMax M2.5 🆕 | 0,3 USD/1 millió bemenet | 5 órás gurulás | Érvelés + ügynöki feladatok | -| | MiniMax M2.1 | 0,2 USD/1M | 5 órás gurulás | Legolcsóbb lehetőség | -| | Kimi K2.5 (Moonshot API) 🆕 | Felhasználásonkénti fizetés | Nincs | Közvetlen Moonshot API hozzáférés | -| | Kimi K2 | 9 USD/hó lakás | 10 millió token/hó | Előrelátható költség | -| **🆓 INGYENES** | Qoder | **0 USD** | Korlátlan | 5 modell korlátlan | -| | Qwen | **0 USD** | Korlátlan | 4 modell korlátlan | -| | Kiro | **0 USD** | Korlátlan | Claude Sonnet/Haiku (AWS Builder) | -| | LongCat Flash-Lite 🆕 | **0 USD**(50 millió tok/nap 🔥) | 1 RPS | A legnagyobb ingyenes kvóta a Földön | -| | Beporzások AI 🆕 | **0 USD**(nincs szükség kulcsra) | 1 rekv/15mp | GPT-5, Claude, DeepSeek, Llama 4 | -| | Cloudflare Workers AI 🆕 | **0 USD**(10 000 neuron/nap) | ~150 ill./nap | 50+ modell, globális élvonal | -| | Scaleway AI 🆕 | **0 USD**(összesen 1 millió token) | Ár korlátozott | EU/GDPR, Qwen3 235B, Llama 70B | > 🆕**Új modellek hozzáadva (2026. március):**Grok-4 Fast család 0,20 USD/0,50 USD/M áron (1143 ms-os benchmark – 30%-kal gyorsabb, mint a Gemini 2.5 Flash), GLM-5 Z.AI-n keresztül 128K kimenettel, MiniMax M2.5 Vc3-on keresztül, KiepSeedk 2.5-ös okfejtéssel. Moonshot közvetlen API. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 0 dolláros kombinált halom – a teljes ingyenes beállítás:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Zéró költség. Never stops coding.**Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Az alábbi modellek**100%-ban ingyenesek, hitelkártya nélkül**. Az OmniRoute automatikus útvonalakat indít közöttük, ha egy kvóta kifogy – kombinálja őket egy feltörhetetlen 0 dolláros kombinációért.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Modell | Előtag | Limit | Rate Limit | -| -------------------- | ------ | ------------- | ---------------------- | -| `claude-szonett-4,5` | `kr/` |**Korlátlan**| Nincs bejelentett napi felső határ | -| `claude-haiku-4,5` | `kr/` |**Korlátlan**| Nincs bejelentett napi felső határ | -| `claude-opus-4,6` | `kr/` |**Korlátlan**| Legújabb Opus via Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) -| Modell | Előtag | Limit | Rate Limit | -| ------------------- | ------ | ------------- | ---------------- | -| `kimi-k2-gondolkodás` | "ha/" |**Korlátlan**| Nincs bejelentett felső határ | -| "qwen3-coder-plus" | "ha/" |**Korlátlan**| Nincs bejelentett felső határ | -| `deepseek-r1` | "ha/" |**Korlátlan**| Nincs bejelentett felső határ | -| `minimax-m2,1` | "ha/" |**Korlátlan**| Nincs bejelentett felső határ | -| "kimi-k2" | "ha/" |**Korlátlan**| Nincs bejelentett felső határ | +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | --------------------- | +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -> Javasolt csatlakozási mód:**Személyes hozzáférési token + `qodercli`**. A böngésző OAuth -> kísérleti és alapértelmezés szerint le van tiltva, hacsak nincsenek beállítva a `QODER_OAUTH_*` környezeti változók.### 🟡 QWEN MODELS (Device Code Auth) +### 🟢 QODER MODELS (Free PAT via qodercli) -| Modell | Előtag | Limit | Rate Limit | -| -------------------- | ------ | ------------- | -------------------- | -| "qwen3-coder-plus" | `qw/` |**Korlátlan**| Nincs bejelentett felső határ | -| `qwen3-coder-flash` | `qw/` |**Korlátlan**| Nincs bejelentett felső határ | -| `qwen3-coder-next` | `qw/` |**Korlátlan**| Nincs bejelentett felső határ | -| "látás-modell" | `qw/` |**Korlátlan**| Multimodális (képek) |### 🟣 GEMINI CLI (Google OAuth) +| Model | Prefix | Limit | Rate Limit | +| ------------------ | ------ | ------------- | --------------- | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -| Modell | Előtag | Limit | Rate Limit | -| ------------------------- | ------ | ---------------------------- | ------------- | -| `gemini-3-flash-preview` | `gc/` |**180 000 tok/hó**+ 1 000/nap | Havi visszaállítás | -| "gemini-2.5-pro" | `gc/` | 180 000/hó (megosztott medence) | Kiváló minőségű |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| Tier | Napi limit | Rate Limit | Megjegyzések | -| ---------- | ------------ | ----------- | ------------------------------------------------------- | -| Ingyenes (fejlesztő) | Nincs token cap |**~40 RPM**| 70+ modell; átállás a tiszta díjhatárokra 2025 közepén | +### 🟡 QWEN MODELS (Device Code Auth) -Népszerű ingyenes modellek: "moonshotai/kimi-k2.5" (Kimi K2.5), "z-ai/glm4.7" (GLM 4.7), "deepseek-ai/deepseek-v3.2" (DeepSeek V3.2), "nvidia/llama-3.3-70b-deepseek"### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | ------------------- | +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Tier | Napi limit | Rate Limit | Megjegyzések | -| ---- | ------------------ | ----------------- | -------------------------------------------- | -| Ingyenes |**1 millió token/nap**| 60K TPM / 30 RPM | A világ leggyorsabb LLM-következtetése; naponta visszaállítja | +### 🟣 GEMINI CLI (Google OAuth) -Ingyenesen elérhető: "llama-3.3-70b", "llama-3.1-8b", "deepseek-r1-distill-llama-70b"### 🔴 GROQ (Free API Key — console.groq.com) +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -| Tier | Napi limit | Rate Limit | Megjegyzések | -| ---- | ------------- | ----------------- | ------------------------------------------ | -| Ingyenes |**14,4K RPD**| 30 ford./perc modellenként | Nincs hitelkártya; 429 limiten, nem terhelik | +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) -Ingyenesen elérhető: "láma-3.3-70b-veratile", "gemma2-9b-it", "mixtral-8x7b", "whisper-large-v3"### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +| Tier | Daily Limit | Rate Limit | Notes | +| ---------- | ------------ | ----------- | ------------------------------------------------------ | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -| Modell | Előtag | Napi ingyenes kvóta | Megjegyzések | -| ------------------------------ | ------ | ------------------ | ------------------------ | -| "LongCat-Flash-Lite" | "lc/" |**50 millió token**💥 | A valaha volt legnagyobb ingyenes kvóta | -| `LongCat-Flash-Chat` | "lc/" | 500 000 token | Többfordulós csevegés | -| "LongCat-Flash-Thinking" | "lc/" | 500 000 token | Érvelés / CoT | -| `LongCat-Flash-Thinking-2601` | "lc/" | 500 000 token | 2026. januári verzió | -| `LongCat-Flash-Omni-2603` | "lc/" | 500 000 token | Multimodális | +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -> 100%-ban ingyenes nyilvános bétaverzióban. Regisztráljon a [longcat.chat](https://longcat.chat) oldalon e-mailben vagy telefonon. Napi alaphelyzetbe állítás 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -| Modell | Előtag | Rate Limit | Szolgáltató mögött | -| ---------- | ------ | ---------- | ------------------- | -| "openai" | `pol/` | 1 rekv/15mp | GPT-5 | -| `claude` | `pol/` | 1 rekv/15mp | Antropikus Claude | -| "gemini" | `pol/` | 1 rekv/15mp | Google Gemini | -| `mélyre törekszik` | `pol/` | 1 rekv/15mp | DeepSeek V3 | -| `láma` | `pol/` | 1 rekv/15mp | Meta Llama 4 Scout | -| "mistral" | `pol/` | 1 rekv/15mp | Mistral AI | +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -> ✨**Zéró súrlódás:**Nincs regisztráció, nincs API-kulcs. Adja hozzá a Pollinations szolgáltatót egy üres kulcsmezővel, és azonnal működik.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -| Tier | Napi neuronok | Egyenértékű használat | Megjegyzések | -| ---- | ------------- | ---------------------------------------- | ------------------------ | -| Ingyenes |**10 000**| ~150 LLM ill / 500s hang / 15K beágyazás | Globális élvonal, 50+ modell | +### 🔴 GROQ (Free API Key — console.groq.com) -Népszerű ingyenes modellek: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (ingyenes hang!), `@cf/qwen/qwen2.5-coder-15b-coder- +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ------------- | ---------------- | ----------------------------------------- | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -> API-token + fiókazonosító szükséges a [dash.cloudflare.com] webhelyről (https://dash.cloudflare.com). Tárolja fiókazonosítóját a szolgáltató beállításaiban.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -| Tier | Ingyenes kvóta | Helyszín | Megjegyzések | -| ---- | ------------- | ------------ | ------------------------------------ | -| Ingyenes |**1M token**| 🇫🇷 Párizs, EU | Nincs szükség hitelkártyára a korlátokon belül | +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 -Ingyenesen elérhető: "qwen3-235b-a22b-instruct-2507" (Qwen3 235B!), "llama-3.1-70b-instruct", "mistral-small-3.2-24b-instruct-2506", "deepseek-v3-032" +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | -> EU/GDPR-kompatibilis. Szerezze be az API-kulcsot a [console.scaleway.com](https://console.scaleway.com) címen. +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. ->**💡 The Ultimate Free Stack (11 szolgáltató, 0 USD örökké):** +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | +| ---------- | ------ | ---------- | ------------------ | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | + +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. + +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 + +| Tier | Daily Neurons | Equivalent Usage | Notes | +| ---- | ------------- | --------------------------------------- | ----------------------- | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | + +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` + +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. + +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | +| ---- | ------------- | ------------ | ----------------------------------- | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | + +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` + +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). + +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Sonnet/Haiku KORLÁTALAN -> Qoder (if/) → kimi-k2-gondolkodás, qwen3-coder-plus, deepseek-r1 UNLIMITED -> LongCat Lite (lc/) → LongCat-Flash-Lite – 50 millió token/nap 🔥 -> Beporzások (pol/) → GPT-5, Claude, DeepSeek, Llama 4 – nincs szükség kulcsra -> Qwen (qw/) → qwen3 kódoló modellek KORLÁTALAN -> Gemini (gemini/) → Gemini 2.5 Flash – 1500 rekv/nap ingyenes -> Cloudflare AI (vö./) → 50+ modell – 10 000 neuron/nap -> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1 millió ingyenes token (EU) -> Groq (groq/) → Llama/Gemma – 14,4 ezer rekv/nap ultragyors -> NVIDIA NIM (nvidia/) → 70+ nyitott modell – 40 RPM örökké -> Cerebrák (cerebras/) → Llama/Qwen a világ leggyorsabb – 1 millió tok/nap -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Bármilyen hang/videó átírása**0 USD-ért**– Deepgram vezet 200 USD ingyenes, AssemblyAI 50 USD tartalék, Groq Whisper korlátlan vészhelyzeti tartalékként. +## 🎙️ Free Transcription Combo -| Szolgáltató | Ingyenes kreditek | Legjobb modell | Rate Limit | -| ------------------ | ----------------------- | --------------------------------------------- | ----------------------------- | -| 🟢**Deepgram**|**200 USD ingyenes**(regisztráció) | `nova-3` — a legjobb pontosság, több mint 30 nyelv | Nincs RPM-korlát az ingyenes krediteknél | -| 🔵**AssemblyAI**|**50 USD ingyenes**(regisztráció) | "univerzális-3-pro" – fejezetek, hangulat, személyazonosításra alkalmas adatok | Nincs RPM-korlát az ingyenes krediteknél | -| 🔴**Groq**|**Örökre ingyenes**| `whisper-large-v3` — OpenAI Whisper | 30 RPM (korlátozott sebesség) | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. -**Javasolt kombináció a `/dashboard/combos'-ban:**``` +| Provider | Free Credits | Best Model | Rate Limit | +| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | + +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Ezután a `/dashboard/media` →**Átírás**lapon: töltsön fel bármilyen audio- vagy videofájlt → válassza ki a kombinált végpontot → kérje le az átírást a támogatott formátumokban.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -Az OmniRoute v2.0 működési platformként készült, nem csak közvetítő proxyként.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Funkció | Mit csinál | -| --------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Grok-4 Fast Family** | xAI modellek 0,20 USD/0,50 USD/M áron – 1143 ms benchmark (30%-kal gyorsabb, mint a Gemini 2.5 Flash) | -| 🧠**GLM-5 a Z.AI-n keresztül** | 128 000 kimeneti kontextus, 0,5 USD/1 millió – a GLM család legújabb zászlóshajója | -| 🔮**MiniMax M2.5** | Érvelés + ügynöki feladatok 0,30 USD/1M áron – jelentős fejlesztés az M2.1-hez képest | -| 🎯**ToolCalling Flag modellenként** | Modellenkénti `toolCalling: igaz/hamis` a rendszerleíró adatbázisban – Az AutoCombo kihagyja az eszközzel nem rendelkező modelleket | -| 🌍**Többnyelvű szándékfelismerés** | PT/ZH/ES/AR kulcsszavak az AutoCombo pontozásban – jobb modellválasztás nem angol nyelvű tartalomhoz | -| 📊**Benchmark-vezérelt tartalékok** | Valódi p95 késés az élő kérések hírcsatornáiból, kombinált pontozásból – Az AutoCombo tanul a tényleges adatokból | -| 🔁**Duplikáció visszavonásának kérése** | Tartalom-kivonat alapú dedup ablak – többügynök biztonságos, megakadályozza az ismétlődő terheléseket | -| 🔌**Pluggable RouterStrategy** | Bővíthető "RouterStrategy" interfész – egyéni útválasztási logika hozzáadása pluginként | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Funkció | Mit csinál | -| ------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Játszótér modell** | Irányítópult oldal bármely modell közvetlen teszteléséhez – szolgáltató/modell/végpont választó, Monaco Editor, adatfolyam, megszakítás, időzítés | -| 🔏**CLI ujjlenyomat-egyeztetés** | Szolgáltatónkénti fejléc/törzs rendezés a natív CLI-aláírásoknak megfelelően – váltson szolgáltatónként a Beállítások > Biztonság menüpontban.**A proxy IP-címe megmarad** | -| 🤝**ACP-támogatás (Agent Client Protocol)** | CLI ügynök felderítés (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 további), folyamat spawner, "/api/acp/agents" végpont | -| 🤖**ACP Agents Dashboard** | Hibakeresés › Ügynökök oldal – 14 ügynökből álló rács telepítési állapottal, verzióval, egyéni ügynök űrlappal bármely CLI-eszközhöz. Az**OpenCode**felhasználók egy „Opencode.json letöltése” gombot kapnak, amely automatikusan létrehoz egy használatra kész konfigurációt az összes elérhető modellhez. | -| 🔧**Egyéni modell `apiFormat` Routing** | Egyéni modellek `apiFormat: "responses"` segítségével most már megfelelően irányítják a Responses API fordítóhoz | -| 🏢**Codex Workspace Isolation** | E-mailenként több Codex-munkaterület – az OAuth megfelelően választja el a kapcsolatokat a munkaterület-azonosító | -| 🔄**Elektronikus automatikus frissítés** | Az asztali alkalmazás ellenőrzi a frissítéseket + automatikus telepítés újraindításkor | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Funkció | Mit csinál | -| ---------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**MCP szerver (25 eszköz)** | IDE/agent eszközök 3 átvitelen keresztül: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 mag + 3 memória + 4 ügyességi eszköz | -| 🤝**A2A szerver (JSON-RPC + SSE)** | Ügynök-ügynök feladatvégrehajtás szinkronizálási és adatfolyam-folyamokkal | -| 🧭**Konszolidált végpontok oldal** | Lapos kezelőoldal Endpoint Proxy, MCP, A2A és API Endpoints lapokkal | -| 🎚️**Szolgáltatás engedélyezése/letiltása kapcsolók** | BE/KI kapcsolók az MCP-hez és az A2A-hoz a beállítások fennmaradásával (alapértelmezett: KI) | -| 🛰️**MCP Runtime Heartbeat** | Valós folyamatállapot (pid, üzemidő, szívverés kora, szállítás, hatókör mód) | -| 📋**MCP Audit Trail** | Szűrhető auditnaplók sikerrel/sikertelenséggel és kulcshozzárendeléssel | -| 🔐**MCP Scope Enforcement** | 10 részletes hatókörű engedély az ellenőrzött szerszám-hozzáféréshez | -| 📡**A2A Task Lifecycle Management** | Feladatok listázása/szűrése, események/műtermékek ellenőrzése, futó feladatok megszakítása | -| 📋**Agent Card Discovery** | `/.well-known/agent.json` az ügyfél automatikus felfedezéséhez | -| 🧪**E2E protokoll tesztkábel** | Valódi MCP SDK + A2A kliens a "test:protocols:e2e" | -| ⚙️**Működési vezérlők** | Váltókombó, rugalmassági profilok alkalmazása, megszakítók alaphelyzetbe állítása egyetlen vezérlőfelületről | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Funkció | Mit csinál | -| --------------------------------------------- | --------------------------------------------------------------------------------------- | ----------------------- | -| 🎯**Intelligens 4-szintű tartalék** | Automatikus útvonal: Előfizetés → API-kulcs → Olcsó → Ingyenes | -| 📊**Valós idejű kvótakövetés** | Élő tokenszám + visszaszámlálás visszaállítása szolgáltatónként | -| 🔄**Formátum fordítás** | OpenAI ↔ Claude ↔ Gemini ↔ Válaszok sémabiztos konverziókkal | -| 👥**Többfiókos támogatás** | Több fiók szolgáltatónként intelligens kiválasztással | -| 🔄**Automatikus token frissítés** | Az OAuth-tokenek automatikusan frissülnek | -| 🎨**Egyéni kombók** | 9 kiegyensúlyozási stratégia + tartalék láncvezérlés | -| 🌐**Wildcard Router** | `szolgáltató/*` dinamikus útválasztás | -| 🧠**A költségvetési szabályozás átgondolása** | Áthaladási, automatikus, egyéni és adaptív érvelési korlátok | -| 🔀**Modell álnevek** | Beépített + egyedi modell alias és migrációs biztonság | -| ⚡**Háttérromlás** | Alacsony prioritású háttérfeladatok irányítása olcsóbb modellek felé | -| 🧪**Feladattudatos intelligens útválasztás** | Modell automatikus kiválasztása tartalomtípus szerint (kódolás/látás/elemzés/összegzés) | -| 🔄**A2A ügynök munkafolyamatok** | Determinisztikus FSM hangszerelő állapotfüggő többlépcsős ügynök-végrehajtáshoz | -| 🔀**Adaptív útválasztás** | Dinamikus stratégia felülbírálása a token mennyisége és a prompt bonyolultsága alapján | -| 🎲**Szolgáltatói sokszínűség** | Shannon entrópia pontozás kiegyenlítő automatikus kombinált forgalomelosztás | -| 💬**Rendszer azonnali befecskendezés** | Következetesen alkalmazott globális viselkedésszabályozás | -| 📄**Responses API-kompatibilitás** | Full `/v1/responses` support for Codex and advanced agentic workflows | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Funkció | Mit csinál | -| -------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**Képgenerálás** | `/v1/images/generations' felhővel és helyi háttérprogramokkal | -| 📐**Beágyazás** | `/v1/embeddings' a kereséshez és a RAG-folyamatokhoz | -| 🎤**Audio átírás** | "/v1/audio/transcriptions" – 7 szolgáltató (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), automatikus nyelvérzékelés, MP4/MP3/WAV támogatás | -| 🔊**Szövegfelolvasó** | "/v1/audio/speech" – 10 szolgáltató (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) helyes hibaüzenetekkel | -| 🎬**Videogeneráció** | `/v1/videos/generations` (ComfyUI + SD WebUI munkafolyamatok) | -| 🎵**Zenegeneráció** | `/v1/music/generations' (ComfyUI munkafolyamatok) | -| 🛡️**Moderálás** | "/v1/moderations" biztonsági ellenőrzések | -| 🔀**Átsorolás** | "/v1/rerank" a relevanciapontozáshoz | -| 🔍**Internetes keresés**🆕 | "/v1/search" – 5 szolgáltató (Serper, Brave, Perplexity, Exa, Tavily), 6500+ ingyenes/hó, automatikus feladatátvétel, gyorsítótár | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Funkció | Mit csinál | -| ---------------------------------------------- | ---------------------------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**Megszakítók** | Modellenkénti kioldás/helyreállítás küszöbérték-vezérlőkkel | -| 🎯**Végpont-tudatos modellek** | Az egyéni modellek deklarálják a támogatott végpontokat + API formátumot | -| 🛡️**Menydörgésellenes csorda** | Mutex + szemafor védelem az újrapróbálkozás/rate eseményeknél | -| 🧠**Szemantikai + aláírás gyorsítótár** | Költség/késleltetés csökkentése két gyorsítótár-réteggel | -| ⚡**Idempotencia kérése** | Megkettőzött védőablak | -| 🔒**TLS ujjlenyomat-hamisítás** | Böngészőszerű TLS-ujjlenyomat –**csökkenti a botfelismerést és a fiók megjelölését** | -| 🔏**CLI ujjlenyomat-egyeztetés** | Megfelel a natív CLI-kérés aláírásainak —**csökkenti a kitiltási kockázatot, miközben megőrzi a proxy IP-címét** | -| 🌐**IP-szűrés** | Engedélyezési lista/blokkolista vezérlés a nyílt telepítésekhez | -| 📊**Szerkeszthető díjkorlátok** | Konfigurálható globális/szolgáltatói szintű korlátozások tartósan | -| 📉**Kecses leépülés** | Többrétegű képességek tartalékai az alapvető átjáróműveletek védelmére | -| 📜**Config Audit Trail** | Diff-alapú változáskövetés, amely megakadályozza a működési eltolódást egyszerű visszaállításokkal | -| ⏳**Provider Health Sync** | Proaktív jogkivonat lejárati figyelése, amely riasztásokat vált ki az engedélyezési hibák előtt | -| 🚪**A kitiltott fiókok automatikus letiltása** | A működő megszakító automatikusan lezárja a véglegesen blokkolt tokenszámlákat | -| 🔑**API-kulcskezelés + hatókör** | A kulcsok biztonságos kiadása/forgatása és a modell/szolgáltató vezérlői | -| 👁️**Hatáskörű API-kulcs felfedése**🆕 | Az API-kulcsok visszaállításának engedélyezése a következőn keresztül: `ALLOW_API_KEY_REVEAL` | -| 🛡️**Védett `/modellek`** | Opcionális hitelesítési kapu és szolgáltatói elrejtés a modellkatalógushoz | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Funkció | Mit csinál | -| ----------------------------------- | ---------------------------------------------------------------------------- | ---------------------------- | -| 📝**Kérés + proxynaplózás** | Teljes kérés/válasz és proxynaplózás | -| 📉**Folyamatos részletes naplók**🆕 | Tisztán rekonstruálja az SSE hasznos adatfolyamokat a felhasználói felületbe | -| 📋**Unified Logs Dashboard** | Kérelem, proxy, audit és konzolnézet egy oldalon | -| 🔍**Telemetria kérése** | p50/p95/p99 késleltetés és nyomkövetési kérelem | -| 🏥**Egészségügyi irányítópult** | Üzemidő, megszakítási állapotok, zárolások, gyorsítótár statisztika | -| 💰**Költségkövetés** | Költségvetési vezérlők és modellenkénti árképzés láthatósága | -| 📈**Analytics vizualizációk** | Modell/szolgáltató használati betekintések és trendnézetek | -| 🧪**Értékelési keret** | Arany készlet tesztelése konfigurálható meccsstratégiákkal | -| 📡**Élő diagnosztika**🆕 | Szemantikus gyorsítótár bypass a pontos kombinált élő teszteléshez | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Funkció | Mit csinál | -| ------------------------------------- | --------------------------------------------------------------------------------------- | --------------------- | -| 🌐**Deploy Anywhere** | Localhost, VPS, Docker, Cloud környezetek | -| 🚇**Cloudflare Tunnel**🆕 | Egykattintásos Quick Tunnel integráció az irányítópultról | -| 🔑**API kulcsmodell szűrése** | Natív /v1/models válasz a hozzárendelt hordozókörnyezeti szerepkörökön keresztül szűrve | -| ⚡**Smart Cache Bypass** | Konfigurálható TTL-heurisztika és kényszerített visszatöltési vezérlők | -| 🔄**Biztonsági mentés/visszaállítás** | Export/import és katasztrófa utáni helyreállítási folyamatok | -| 🧙**Bevezető varázsló** | Első futtatás irányított beállítás | -| 🔧**CLI Tools Dashboard** | Egykattintásos beállítás a népszerű kódolóeszközökhöz | -| 🎮**Játszótér modell** | Teszteljen bármely szolgáltatót/modellt/végpontot az irányítópultról | -| 🔏**CLI ujjlenyomat kapcsoló** | Szolgáltatónkénti ujjlenyomat-egyeztetés a Beállítások > Biztonság | -| 🌐**i18n (30 nyelv)** | Teljes irányítópult + dokumentumok nyelvi támogatása RTL lefedettséggel | -| 🧹**Minden modell törlése** | Egykattintásos modelllista törlése a szolgáltató adatai között | -| 👁️**Sidebar Controls**🆕 | Összetevők és integrációk elrejtése a Megjelenés beállításaiból | -| 📋**Kiadássablonok** | Szabványos GitHub-sablonok hibákhoz és szolgáltatásokhoz | -| 📂**Egyéni adattár** | `DATA_DIR` felülírása a tárolási helyhez | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1295,105 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -Ha a kvóta, arány vagy állapot meghiúsul, az OmniRoute kézi váltás nélkül automatikusan a következő jelöltre lép.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- Az MCP + A2A felfedezhető a felhasználói felületen és a dokumentumokban (nem rejtett) -- A protokollállapot API-k élő működési adatokat tesznek közzé (`/api/mcp/*`, `/api/a2a/*`) -- Az irányítópultok tartalmaznak műveleteket a 2. napi műveletekhez (kombinált kapcsolók, megszakítók alaphelyzetbe állítása, feladat törlése)#### Translator + validation workflow +#### Protocol management that is visible and operable -A Fordító terület a következőket tartalmazza: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Játszótér**: kérjen átalakítási ellenőrzéseket -**Csevegés tesztelő**: teljes kérés/válasz oda-vissza út -**Tesztpad**: több eset egy menetben -**Élő monitor**: valós idejű forgalmi nézet +#### Translator + validation workflow -Plusz a protokoll érvényesítése valós kliensekkel az "npm run test:protocols:e2e" segítségével. +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Eszközreferencia, IDE-konfigurációk és példák kliensekre +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A kiszolgáló README](src/lib/a2a/README.md)**– Készségek, JSON-RPC metódusok, adatfolyamok és feladatok életciklusa## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -Az OmniRoute egy beépített kiértékelő keretrendszert tartalmaz, amellyel az LLM válaszminőségét egy aranykészlettel összehasonlítva tesztelheti. Az irányítópulton az**Analytics → Evals**menüpontban érheti el.### Built-in Golden Set +## 🧪 Evaluations (Evals) -Az előre feltöltött "OmniRoute Golden Set" teszteseteket tartalmaz: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Üdvözlet, matematika, földrajz, kódgenerálás -- JSON formátum megfelelőség, fordítás, leértékelés generálása -- Biztonsági elutasítás (káros tartalom), számlálás, logikai logika### Evaluation Strategies +### Built-in Golden Set -| Stratégia | Leírás | Példa | -| ----------- | ------------------------------------------------------------------------------------------------- | --------------------------------- | --- | -| "pontos" | A kimenetnek pontosan meg kell egyeznie | "4" | -| `tartalmaz` | A kimenetnek tartalmaznia kell részkarakterláncot (a kis- és nagybetűk nem különböznek egymástól) | "Párizs" | -| "regex" | A kimenetnek meg kell egyeznie a regex mintával | "1.*2.*3" | -| "egyedi" | Az egyéni JS függvény igaz/hamis | `(kimenet) => output.length > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) - +
+🧩 MCP Setup (Model Context Protocol) -🧩 MCP beállítása (Model Context Protocol) +Start MCP transport in stdio mode: -Indítsa el az MCP-átvitelt stdio módban:```bash +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Javasolt érvényesítési folyamat: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Csatlakoztassa az MCP-klienst az stdio-n keresztül. -2. Futtassa az „omniroute_get_health” parancsot. -3. Futtassa az `omniroute_list_combos` parancsot. -4. Nyissa meg a „/dashboard/mcp” mappát a szívverés, a tevékenység és az ellenőrzés megerősítéséhez. - -Hasznos API-k az automatizáláshoz: +Useful APIs for automation: - `GET /api/mcp/status` - `GET /api/mcp/tools` - `GET /api/mcp/audit` -- "GET /api/mcp/audit/stats".
+- `GET /api/mcp/audit/stats` - -🤝 A2A beállítás (Agent2Agent) + -Fedezze fel az ügynököt:```bash +
+🤝 A2A Setup (Agent2Agent) + +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Feladat küldése:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` - -Életciklus kezelése: +Manage lifecycle: - `GET /api/a2a/status` - `GET /api/a2a/tasks` - `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -Működési UI: +Operational UI: -- `/dashboard/a2a` a feladat/állapot/folyam megfigyelhetőségéhez és füstműveletekhez
+- `/dashboard/a2a` for task/state/stream observability and smoke actions - -🧪 Végpontok közötti protokollellenőrzés + -Érvényesítse mindkét protokollt valódi ügyfelekkel:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Ez igazolja: +This verifies: -- MCP SDK kliens csatlakozás/lista/hívás -- A2A felfedezés/küldés/stream/get/cancel -- Az MCP audit és A2A feladatkezelő API-k adatainak keresztellenőrzése
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs - + -💳 Előfizetéses szolgáltatók### Claude Code (Pro/Max) +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1406,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Profi tipp:**Használja az Opust összetett feladatokhoz, a Sonnet pedig a sebességhez. Az OmniRoute nyomkövetési kvóta modellenként!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1420,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Mostantól minden Codex-fiók rendelkezik házirend-kapcsolókkal az "Irányítópult -> Szolgáltatók" részben: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- "5h" (BE/KI): az 5 órás ablak küszöbszabályának érvényesítése. -- `Heti` (BE/KI): érvényesíti a heti ablak küszöbértékét. -- Küszöbbeli viselkedés: ha egy engedélyezett ablak eléri a >=90%-os használatot, a fiók kimarad. -- Forgatási viselkedés: Az OmniRoute automatikusan a következő jogosult Codex-fiókhoz irányít. -- Visszaállítási viselkedés: amikor a szolgáltató „resetAt” ideje letelik, a fiók automatikusan újra jogosulttá válik. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Forgatókönyvek: +Scenarios: -- `5h ON` + `Heti BE`: a fiók kimarad, ha valamelyik ablak eléri a küszöbértéket. -- `5h OFF` + `Heti BE`: csak heti használat blokkolhatja a fiókot. -- `5h BE` + `Heti KI`: csak 5 órás használat blokkolhatja a fiókot. -- `resetAt` sikeres: a fiók automatikusan újraindul a forgatásba (nincs kézi újraengedélyezés).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1445,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Legjobb érték:**Hatalmas ingyenes szint! Használja ezt a fizetett szintek előtt.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1460,74 +1662,91 @@ Models:
- +
+🔑 API Key Providers -🔑 API-kulcsszolgáltatók### NVIDIA NIM (FREE developer access — 70+ models) +### NVIDIA NIM (FREE developer access — 70+ models) -1. Regisztráljon: [build.nvidia.com](https://build.nvidia.com) -2. Ingyenes API-kulcs beszerzése (1000 következtetési kredit) -3. Irányítópult → Szolgáltató hozzáadása → NVIDIA NIM: - - API-kulcs: "nvapi-your-key". +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Modelek:**"nvidia/llama-3.3-70b-instruct", "nvidia/mistral-7b-instruct" és több mint 50 további +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -**Profi tipp:**OpenAI-kompatibilis API – zökkenőmentesen működik az OmniRoute formátumfordításával!### DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -1. Regisztráljon: [platform.deepseek.com](https://platform.deepseek.com) -2. Szerezze be az API-kulcsot -3. Irányítópult → Szolgáltató hozzáadása → DeepSeek +### DeepSeek -**Modellek:**"deepseek/deepseek-chat", "deepseek/deepseek-coder"### Groq (Free Tier Available!) +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -1. Regisztráljon: [console.groq.com](https://console.groq.com) -2. API-kulcs beszerzése (ingyenes szint tartalmazza) -3. Irányítópult → Szolgáltató hozzáadása → Groq +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Modellek:**"groq/llama-3.3-70b", "groq/mixtral-8x7b" +### Groq (Free Tier Available!) -**Profi tipp:**Ultragyors következtetés – a legjobb valós idejű kódoláshoz!### OpenRouter (100+ Models) +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -1. Regisztráljon: [openrouter.ai](https://openrouter.ai) -2. Szerezze be az API-kulcsot -3. Irányítópult → Szolgáltató hozzáadása → OpenRouter +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Modellek:**Hozzáférés több mint 100 modellhez az összes főbb szolgáltatótól egyetlen API-kulccsal. +**Pro Tip:** Ultra-fast inference — best for real-time coding! -**Az irányítópult viselkedése:**Az OpenRouter modellek kezelése az**Elérhető modellek**oldalon történik. A kézi hozzáadása, importálása és automatikus szinkronizálása ugyanazt a listát frissíti.
+### OpenRouter (100+ Models) - +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -💰 Olcsó szolgáltatók (tartalék)### GLM-4.7 (Daily reset, $0.6/1M) +**Models:** Access 100+ models from all major providers through a single API key. -1. Regisztráljon: [Zhipu AI](https://open.bigmodel.cn/) -2. Szerezze be az API-kulcsot a Coding Plan-ból -3. Irányítópult → API-kulcs hozzáadása: - - Szolgáltató: "glm". - - API-kulcs: "a-kulcs". +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -**Használja:**`glm/glm-4.7` + -**Profi tipp:**A kódolási terv 3-szoros kvótát kínál 1/7 költséggel! Visszaállítás naponta 10:00.### MiniMax M2.1 (5h reset, $0.20/1M) +
+💰 Cheap Providers (Backup) -1. Regisztráljon: [MiniMax](https://www.minimax.io/) -2. Szerezze be az API-kulcsot -3. Irányítópult → API-kulcs hozzáadása +### GLM-4.7 (Daily reset, $0.6/1M) -**Használja:**`minimax/MiniMax-M2.1` +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**Profi tipp:**A legolcsóbb lehetőség hosszú kontextushoz (1 millió token)!### Kimi K2 ($9/month flat) +**Use:** `glm/glm-4.7` -1. Feliratkozás: [Moonshot AI](https://platform.moonshot.ai/) -2. Szerezze be az API-kulcsot -3. Irányítópult → API-kulcs hozzáadása +**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. -**Használd:**`kimi/kimi-latest` +### MiniMax M2.1 (5h reset, $0.20/1M) -**Profi tipp:**Fix 9 USD/hó 10 millió token esetén = 0,90 USD/1 millió tényleges költség!
+1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key - +**Use:** `minimax/MiniMax-M2.1` -🆓 INGYENES szolgáltatók (vészhelyzeti biztonsági mentés)### Qoder (5 FREE models via OAuth) +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1568,9 +1787,10 @@ Models:
- +
+🎨 Create Combos -🎨 Hozzon létre kombókat### Example 1: Maximize Subscription → Cheap Backup +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1598,9 +1818,10 @@ Cost: $0 forever!
- +
+🔧 CLI Integration -🔧 CLI-integráció### Cursor IDE +### Cursor IDE ``` Settings → Models → Advanced: @@ -1611,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Használja az irányítópult**CLI Tools**oldalát az egykattintásos konfiguráláshoz, vagy szerkessze manuálisan a `~/.claude/settings.json` fájlt.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1622,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**1. lehetőség – Irányítópult (ajánlott):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**2. lehetőség – Kézi:**Az `~/.openclaw/openclaw.json` szerkesztése:```json +```json { "models": { "providers": { @@ -1639,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Megjegyzés:**Az OpenClaw csak a helyi OmniRoute-tal működik. Az IPv6-feloldási problémák elkerülése érdekében használja a "127.0.0.1" értéket a "localhost" helyett.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1653,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**1. lépés:**Az OmniRoute hozzáadása egyéni szolgáltatóként:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**2. lépés:**Hozza létre/szerkesztse az `opencode.json` fájlt a projekt gyökérjében:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1679,117 +1909,130 @@ opencode } } } -```` +``` -**3. lépés:**Válassza ki a modellt az OpenCode-ban:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Tipp:**Adjon hozzá bármely, az OmniRoute `/v1/models' végpontjában elérhető modellt a `modellek' szakaszhoz. Használja a „szolgáltató/modellazonosító” formátumot az OmniRoute irányítópultján.
+ --- ## Hibaelhárítás - -Kattintson ide a hibaelhárítási útmutató kibontásához +
+Click to expand troubleshooting guide -**"A nyelvi modell nem adott üzenetet"** +**"Language model did not provide messages"** -- A szolgáltatói kvóta kimerült → Ellenőrizze az irányítópult kvótakövetőjét -- Megoldás: Használjon kombinált tartalékot, vagy váltson olcsóbb szintre +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**Drátakorlát** +**Rate limiting** -- Előfizetési kvóta lejárt → Tartalék a GLM/MiniMax-hoz -- Kombó hozzáadása: "cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking" +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**OAuth token lejárt** +**OAuth token expired** -- Az OmniRoute automatikusan frissíti -- Ha a problémák továbbra is fennállnak: Irányítópult → Szolgáltató → Újracsatlakozás +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Magas költségek** +**High costs** -- Ellenőrizze a használati statisztikákat az Irányítópult → Költségek menüpontban -- Állítsa át az elsődleges modellt GLM/MiniMax-ra -- Használjon ingyenes réteget (Gemini CLI, Qoder) a nem kritikus feladatokhoz +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Az irányítópult/API portok hibásak** +**Dashboard/API ports are wrong** -- A "PORT" a kanonikus alapport (és alapértelmezés szerint API-port) -- Az „API_PORT” csak az OpenAI-kompatibilis API figyelőt írja felül -- A „DASHBOARD_PORT” csak az irányítópultot/Next.js figyelőt írja felül -- Állítsa be a „NEXT_PUBLIC_BASE_URL” címet irányítópultjára/nyilvános URL-címére (OAuth-visszahívásokhoz) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Felhő szinkronizálási hibák** +**Cloud sync errors** -- Ellenőrizze, hogy a "BASE_URL" a futó példányra mutat -- Ellenőrizze, hogy a „CLOUD_URL” a várható felhő-végpontra mutat -- Tartsa a `NEXT_PUBLIC_*` értékeket a szerveroldali értékekkel összhangban +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Az első bejelentkezés nem működik** +**First login not working** -- Ellenőrizze az "INITIAL_PASSWORD" értéket a ".env" fájlban -- Ha nincs beállítva, a tartalék jelszó `123456` +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Nincs kérésnapló** +**No request logs** -- A kérés melléktermékei kérésenként egy JSON-fájlként íródnak a `DATA_DIR/call_logs/` mappába -- Engedélyezze a folyamat rögzítését az Irányítópult → Naplók → Kérelemnaplók menüpontból, ha részletes, szakaszonkénti hasznos adatokra van szüksége -- Állítsa be az "APP_LOG_TO_FILE=true" értéket, ha az alkalmazáskonzolnaplókat is szeretné a "logs/application/app.log" fájlban -- Szükség szerint állítsa be a `APP_LOG_MAX_FILE_SIZE', 'APP_LOG_RETENTION_DAYS', 'APP_LOG_MAX_FILES' és 'CALL_LOG_MAX_ENTRIES' +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**A csatlakozási teszt „Érvénytelen” üzenetet mutat az OpenAI-kompatibilis szolgáltatók esetében** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Sok szolgáltató nem tesz közzé „/models” végpontot -- Az OmniRoute v1.0.6+ tartalmazza a tartalék érvényesítést a csevegés befejezésén keresztül -- Győződjön meg arról, hogy az alap URL tartalmazza a „/v1” utótagot### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ Fontos az OmniRoute-ot VPS-en, Dockeren vagy bármely távoli szerveren futtató felhasználók számára**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -Az**Antigravity**és**Gemini CLI**szolgáltatók a**Google OAuth 2.0**szolgáltatást használják. A Google megköveteli, hogy az OAuth-folyamatban szereplő „redirect_uri” pontosan egyezzen az alkalmazás Google Cloud Console-jában előregisztrált URI-k egyikével. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -Az OmniRoute csomagban található OAuth hitelesítő adatok**csak a „localhost” számára vannak regisztrálva**. Amikor egy távoli szerveren éri el az OmniRoute-ot (pl. `https://omniroute.myserver.com`), a Google a következőkkel utasítja el a hitelesítést:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Létre kell hoznia egy**OAuth 2.0 ügyfél-azonosítót**a Google Cloud Console-ban a szerver URI-jával.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Nyissa meg a Google Cloud Console-t** +#### Step-by-step -Keresse fel a következőt: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Új OAuth 2.0 ügyfél-azonosító létrehozása** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Kattintson a**„+ Hitelesítési adatok létrehozása”**→**„OAuth-ügyfélazonosító”**elemre. -- Alkalmazás típusa:**"Web alkalmazás"** -- Név: bármi, ami tetszik (pl. "OmniRoute Remote") +**2. Create a new OAuth 2.0 Client ID** -**3. Engedélyezett átirányítási URI-k hozzáadása** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -Az**"Engedélyezett átirányítási URI-k"**mezőbe írja be:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Cserélje ki a "your-server.com" címet a szerver domainjére vagy IP-címére (ha szükséges, adja meg a portot, pl. "http://45.33.32.156:20128/callback"). +**4. Save and copy the credentials** -**4. Mentse és másolja a hitelesítő adatokat** +After creating, Google will show the **Client ID** and **Client Secret**. -A létrehozás után a Google megjeleníti az**Client ID**és**Client Secret**kódot. +**5. Set environment variables** -**5. Környezeti változók beállítása** +In your `.env` (or Docker environment variables): -Az „.env” (vagy a Docker környezeti változókban):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1798,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Az OmniRoute újraindítása**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Próbáljon újra csatlakozni** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Irányítópult → Szolgáltatók → Antigravity (vagy Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -A Google most megfelelően átirányítja a `https://your-server.com/callback` címre.--- +--- #### Temporary workaround (without custom credentials) -Ha most nem szeretné beállítani a saját hitelesítő adatait, továbbra is használhatja a**manuális URL-folyamatot**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. Az OmniRoute megnyitja a Google engedélyezési URL-címét +1. OmniRoute opens the Google authorization URL 2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) -3.**Másolja ki a teljes URL-t**a böngésző címsorából (még akkor is, ha az oldal nem töltődik be) -4. Illessze be az URL-t az OmniRoute csatlakozási módban látható mezőbe -5. Kattintson a**"Csatlakozás"**gombra. +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Ez azért működik, mert az URL-ben szereplő engedélyezési kód attól függetlenül érvényes, hogy az átirányítási oldal betöltődött-e.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. - -🇧🇷 Versão em Português#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Az**Antigravitáció**és a**Gemini CLI**usam**Google OAuth 2.0**hitelesítése. O Google exige que a `redirect_uri` usada no fluxo OAuth seja**exatamente**uma das URI-k pre-cadastradas no Google Cloud Console do aplicativo. +
+🇧🇷 Versão em Português -As credenciais OAuth embutidas no OmniRoute estão cadastradas**apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (pl.: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Você precisa criar um**OAuth 2.0 ügyfél-azonosító**nincs Google Cloud Console com egy URI do seu servidor.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. + +#### Passo a passo **1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -**2. Crie um novo OAuth 2.0 ügyfél-azonosító** +**2. Crie um novo OAuth 2.0 Client ID** -- Kattintson a gombra**"+ Hitelesítési adatok létrehozása"**→**"OAuth-kliens-azonosító"** -- Tipo de Aplicativo:**"Web alkalmazás"** -- Név: escolha qualquer nome (pl.: "OmniRoute Remote") +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Adicione mint engedélyezett átirányítási URI** +**3. Adicione as Authorized Redirect URIs** -No campo**"Engedélyezett átirányítási URI-k"**, kiegészítés:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Helyettesítse a "seu-servidor.com" pelo domínio vagy IP do seu servidor címet (beleértve a porta se necessário-t is, pl.: "http://45.33.32.156:20128/callback"). +**4. Salve e copie as credenciais** -**4. Másolat mentése hitelesítésként** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Após criar, o Google mostrará o**Client ID**e o**Client Secret**. +**5. Configure as variáveis de ambiente** -**5. Konfigurálás variáveis de ambienteként** +No seu `.env` (ou nas variáveis de ambiente do Docker): -No seu `.env` (ou nas variáveis de ambiente do Docker):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1877,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reinicie o OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute - -```` +``` **7. Tente conectar novamente** -Irányítópult → Szolgáltatók → Antigravity (vagy Gemini CLI) → OAuth +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` és autenticação funcionará.--- +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. + +--- #### Workaround temporário (sem configurar credenciais próprias) -Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo**manual de URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. OmniRoute abrirá a Google autorização URL-jét -2. Após você autorizar, o Google tentará redirecionar para "localhost" (que falha no servidor remoto) -3.**Teljes URL másolása**da barra de endereço do seu browser (mesmo que a página não carregue) +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) 4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute -5. Kattintson a**"Connect"**gombra +5. Clique em **"Connect"** -> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1915,64 +2171,73 @@ Se não quiser criar credenciais próprias agora, ainda é possível usar o flux ## 🛠️ Tech Stack - -Kattintson ide a technológiai verem részleteinek kibontásához +
+Click to expand tech stack details --**Futtatási idejű**: Node.js 18–22 LTS (⚠️ A Node.js 24+**nem támogatott**– a `better-sqlite3` natív binárisok nem kompatibilisek) --**Nyelv**: TypeScript 5.9 –**100% TypeScript**az „src/” és az „open-sse/” protokollokon keresztül (nulla „bármilyen” az alapmodulokban a v2.0 óta) --**Keretrendszer**: Next.js 16 + React 19 + Tailwind CSS 4 --**Adatbázis**: LowDB (JSON) + SQLite (tartomány állapota + proxynaplók + MCP-audit + útválasztási döntések) --**Sémák**: Zod (MCP-eszköz I/O-ellenőrzése, API-szerződések) --**Protokollok**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Streaming**: Szerver által küldött események (SSE) --**Auth**: OAuth 2.0 (PKCE) + JWT + API-kulcsok + MCP-hatókörű engedélyezés --**Tesztelés**: Node.js tesztfutó + Vitest (900+ teszt, beleértve az egységet, az integrációt, az E2E-t) --**CI/CD**: GitHub Actions (automatikus npm közzététel + Docker Hub kiadáskor) --**Webhely**: [omniroute.online](https://omniroute.online) --**Csomag**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Rugalmasság**: megszakító, exponenciális visszakapcsolás, mennydörgés elleni csorda, TLS-hamisítás, automatikus kombinált öngyógyítás
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Dokumentáció -| dokumentum | Leírás | -| ----------------------------------------------- | ---------------------------------------------------- | -| [Felhasználói útmutató](docs/USER_GUIDE.md) | Szolgáltatók, kombók, CLI-integráció, telepítés | -| [API-referencia](docs/API_REFERENCE.md) | Minden végpont példákkal | -| [MCP-kiszolgáló](open-sse/mcp-server/README.md) | 16 MCP-eszköz, IDE konfigurációk, Python/TS/Go kliensek | -| [A2A szerver](src/lib/a2a/README.md) | JSON-RPC 2.0 protokoll, készségek, adatfolyam, feladat mgmt | -| [Auto-Combo Engine](docs/auto-combo.md) | 6 faktoros pontozás, módcsomagok, öngyógyító | -| [Hibaelhárítás](docs/TROUBLESHOOTING.md) | Gyakori problémák és megoldások | -| [Architektúra](docs/ARCHITECTURE.md) | Rendszerarchitektúra és belső elemek | -| [Hozzájárulás](CONTRIBUTING.md) | Fejlesztési beállítások és irányelvek | -| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specifikáció | -| [Biztonsági politika](SECURITY.md) | Sebezhetőségi jelentések és biztonsági gyakorlatok | -| [VM-telepítés](docs/VM_DEPLOYMENT_GUIDE.md) | Teljes útmutató: VM + nginx + Cloudflare beállítás | -| [Features Gallery](docs/FEATURES.md) | Vizuális irányítópult bemutató képernyőképekkel | -| [Kiadási ellenőrzőlista](docs/RELEASE_CHECKLIST.md) | Kiadás előtti érvényesítési lépések |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -Az OmniRoute**210+ funkciót tervez**több fejlesztési fázisban. Íme a legfontosabb területek: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Kategória | Tervezett funkciók | Kiemelések | -| ------------------------------ | ----------------- | -------------------------------------------------------------------------------------- | -| 🧠**Útválasztás és intelligencia**| 25+ | Legkisebb késleltetésű útválasztás, címke alapú útválasztás, kvóta elővizsgálat, P2C-fiók kiválasztása | -| 🔒**Biztonság és megfelelőség**| 20+ | SSRF keményítés, hitelesítő adatok álcázása, végpontonkénti sebességkorlát, felügyeleti kulcs hatóköre | -| 📊**Megfigyelhetőség**| 15+ | OpenTelemetry integráció, valós idejű kvótafigyelés, modellenkénti költségkövetés | -| 🔄**Szolgáltatói integrációk**| 20+ | Dinamikus modellnyilvántartás, szolgáltatói leállások, többfiókos Codex, másodpilóta kvótaelemzés | -| ⚡**Teljesítmény**| 15+ | Kettős gyorsítótárréteg, gyorsítótár, válaszgyorsítótár, folyamatos adatfolyam, kötegelt API | -| 🌐**Ökoszisztéma**| 10+ | WebSocket API, config hot-reload, elosztott konfigurációs tároló, kereskedelmi mód |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**OpenCode integráció**- Natív szolgáltatói támogatás az OpenCode AI kódoló IDE-hez -- 🔗**TRAE integráció**— A TRAE AI fejlesztési keret teljes támogatása -- 📦**Batch API**- Aszinkron kötegelt feldolgozás tömeges kérésekhez -- 🎯**Címke alapú útválasztás**- Egyéni címkéken és metaadatokon alapuló útvonalkérések -- 💰**Legalacsonyabb költségű stratégia**- Automatikusan válassza ki a legolcsóbb elérhető szolgáltatót +### 🔜 Coming Soon -> 📝 A teljes funkcióspecifikáció elérhető a [`docs/new-features/`](docs/new-features/) oldalon (217 részletes specifikáció)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1980,18 +2245,20 @@ Az OmniRoute**210+ funkciót tervez**több fejlesztési fázisban. Íme a legfon ### How to Contribute -1. Fork a tároló -2. Hozza létre a szolgáltatási ágat (`git checkout -b feature/amazing-feature`) -3. Végezze el a változtatásokat (`git commit -m 'Elképesztő funkció hozzáadása'`) -4. Nyomja le az ágra (`git push origin funkció/csodálatos szolgáltatás`) -5. Nyisson meg egy lehívási kérelmet +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -A részletes útmutatásért lásd: [CONTRIBUTING.md](CONTRIBUTING.md).### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -2003,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Külön köszönet a**[decolua](https://github.com/decolua)\*\***[9router](https://github.com/decolua/9router)\*\*-nak – az eredeti projektnek, amely ezt a villát inspirálta. Az OmniRoute erre a hihetetlen alapra épít további funkciókkal, multimodális API-kkal és teljes TypeScript-újraírással. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Külön köszönet a**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**-nak – az eredeti Go implementációnak, amely ihlette ezt a JavaScript-portot.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Licenc -MIT-licenc – részletekért lásd: [LICENSE](LICENSE).--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/hu/docs/ARCHITECTURE.md b/docs/i18n/hu/docs/ARCHITECTURE.md index f659c6ecb4..aab37b02ed 100644 --- a/docs/i18n/hu/docs/ARCHITECTURE.md +++ b/docs/i18n/hu/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Utolsó frissítés: 2026-03-28_## Executive Summary -Az OmniRoute egy helyi mesterséges intelligencia-útválasztó átjáró és irányítópult, amely a Next.js-re épül. -Egyetlen OpenAI-kompatibilis végpontot (`/v1/*`) biztosít, és a forgalmat több upstream szolgáltató között irányítja át fordítással, tartalékkal, tokenfrissítéssel és használati követéssel. -Alapvető képességek: +_Last updated: 2026-03-28_ -- OpenAI-kompatibilis API felület a CLI-hez/eszközökhöz (28 szolgáltató) -- Fordítás kérése/válaszolása a szolgáltatói formátumok között -- Model kombinált tartalék (több modell sorozat) -- Fiókszintű tartalék (szolgáltatónként több fiók) -- OAuth + API-kulcs szolgáltatói kapcsolatkezelés -- Beágyazás generálása a „/v1/embeddings” fájlon keresztül (6 szolgáltató, 9 modell) -- Képgenerálás a `/v1/images/generations' fájlon keresztül (4 szolgáltató, 9 modell) -- Gondoljon a címkeelemzésre (`...`) az érvelési modellekhez -- Válasz fertőtlenítés a szigorú OpenAI SDK-kompatibilitás érdekében -- Szerepnormalizálás (fejlesztő→rendszer, rendszer→felhasználó) a szolgáltatók közötti kompatibilitás érdekében -- Strukturált kimenet átalakítás (json_schema → Gemini responseSchema) -- Helyi kitartás a szolgáltatók, kulcsok, álnevek, kombinációk, beállítások, árképzés számára -- Használat/költségkövetés és kérések naplózása -- Opcionális felhőszinkronizálás több eszköz/állapot szinkronizáláshoz -- IP engedélyezési/blokkolási lista API hozzáférés-vezérléshez -- Átgondolt költségvetés-kezelés (áthaladó/automatikus/egyéni/adaptív) -- Globális rendszer azonnali befecskendezése -- Munkamenet követés és ujjlenyomat -- Fiókonként továbbfejlesztett díjkorlátozás szolgáltató-specifikus profilokkal -- Megszakító minta a szolgáltatói rugalmasság érdekében -- Mennydörgés elleni állományvédelem mutex zárral -- Aláírás alapú kérés deduplikációs gyorsítótár -- Domain réteg: modell elérhetősége, költségszabályok, tartalék házirend, kizárási szabályzat -- Tartomány állapotának fennmaradása (SQLite átírási gyorsítótár tartalékok, költségvetések, zárolások, megszakítók számára) -- Házirend motor a kérelmek központosított értékeléséhez (zárás → költségvetés → tartalék) -- Telemetria kérése p50/p95/p99 késleltetési összesítéssel -- Korrelációs azonosító (X-Request-Id) a végpontok közötti nyomkövetéshez -- Megfelelőségi naplózás API-kulcsonkénti leiratkozással -- Eval keretrendszer az LLM minőségbiztosításhoz -- Rugalmas UI műszerfal valós idejű megszakító állapottal -- Moduláris OAuth-szolgáltatók (12 különálló modul az `src/lib/oauth/providers/` alatt) +## Executive Summary -Elsődleges futásidejű modell: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- A Next.js alkalmazásútvonalai az `src/app/api/*` alatt mind az irányítópult API-kat, mind a kompatibilitási API-kat megvalósítják -- Egy megosztott SSE/routing mag az `src/sse/*` + `open-sse/*` állományban kezeli a szolgáltató végrehajtását, fordítását, adatfolyamát, tartalékát és használatát## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Helyi átjáró futásidejű -- Irányítópult-kezelő API-k -- Szolgáltató hitelesítése és token frissítése -- Fordítás és SSE streaming kérése -- Helyi állapot + használat tartóssága -- Opcionális felhőszinkronizálás### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Felhőszolgáltatás megvalósítása a `NEXT_PUBLIC_CLOUD_URL' mögött -- Szolgáltató SLA/vezérlő síkja a helyi folyamaton kívül -- Maguk a külső CLI binárisok (Claude CLI, Codex CLI stb.)## Dashboard Surface (Current) +### Out of Scope -Főoldalak az `src/app/(dashboard)/dashboard/` alatt: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — gyorsindítás + szolgáltató áttekintése -- "/dashboard/endpoint" - végpont proxy + MCP + A2A + API végpont lapjai -- "/dashboard/providers" – szolgáltatói kapcsolatok és hitelesítő adatok -- "/dashboard/combos" - kombinált stratégiák, sablonok, modell-útválasztási szabályok -- "/dashboard/costs" – a költségek összesítése és az árképzés láthatósága -- "/dashboard/analytics" — használati elemzések és kiértékelések -- "/dashboard/limits" – kvóta/kamat szabályozás -- "/dashboard/cli-tools" - CLI-beépítés, futásidejű észlelés, konfiguráció generálása -- "/dashboard/agents" — észlelt ACP ügynökök + egyéni ügynök regisztráció -- `/dashboard/media` — kép/videó/zene játszótér -- "/dashboard/search-tools" – a keresőszolgáltató tesztelése és előzményei -- "/dashboard/health" – üzemidő, megszakítók, sebességkorlátok -- "/dashboard/logs" — kérés/proxy/audit/konzolnaplók -- "/dashboard/settings" – rendszerbeállítások lapjai (általános, útválasztás, kombinált alapértelmezések stb.) -- `/dashboard/api-manager` – API kulcs életciklusa és modellengedélyei## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Fő könyvtárak: +Main directories: -- `src/app/api/v1/*` és `src/app/api/v1beta/*` a kompatibilitási API-khoz -- `src/app/api/*` a felügyeleti/konfigurációs API-khoz -- Következő átírja a `next.config.mjs` `/v1/*` leképezését `/api/v1/*`-re +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Fontos kompatibilitási útvonalak: +Important compatibility routes: -- "src/app/api/v1/chat/completions/route.ts". +- `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- "src/app/api/v1/models/route.ts" - egyéni modelleket tartalmaz "custom: true" -- "src/app/api/v1/embeddings/route.ts" - beágyazás generálása (6 szolgáltató) -- "src/app/api/v1/images/generations/route.ts" - képgenerálás (4+ szolgáltató, beleértve az Antigravitációt/Nebiust) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- "src/app/api/v1/providers/[provider]/chat/completions/route.ts" – dedikált szolgáltatónkénti csevegés -- "src/app/api/v1/providers/[szolgáltató]/embeddings/route.ts" – dedikált szolgáltatónkénti beágyazások -- "src/app/api/v1/providers/[szolgáltató]/images/generations/route.ts" – szolgáltatónként dedikált képek +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` -- `src/app/api/v1beta/models/[...útvonal]/route.ts` +- `src/app/api/v1beta/models/[...path]/route.ts` -Kezelési tartományok: +Management domains: -- Hitelesítés/beállítások: `src/app/api/auth/*`, `src/app/api/settings/*` -- Szolgáltatók/kapcsolatok: `src/app/api/providers*` -- Szolgáltatói csomópontok: `src/app/api/provider-nodes*` -- Egyéni modellek: "src/app/api/provider-models" (GET/POST/DELETE) -- Modellkatalógus: `src/app/api/models/route.ts` (GET) -- Proxy konfigurációja: "src/app/api/settings/proxy" (GET/PUT/DELETE) + "src/app/api/settings/proxy/test" (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- Keys/aliases/combos/pricing: "src/app/api/keys*", "src/app/api/models/alias", "src/app/api/combos*", "src/app/api/pricing" -- Használat: `src/app/api/usage/*` -- Szinkronizálás/felhő: `src/app/api/sync/*`, `src/app/api/cloud/*` -- CLI-eszközök segédei: `src/app/api/cli-tools/*` -- IP-szűrő: "src/app/api/settings/ip-filter" (GET/PUT) -- Gondolkodási költségkeret: `src/app/api/settings/thinking-budget' (GET/PUT) -- Rendszerprompt: "src/app/api/settings/system-prompt" (GET/PUT) -- Munkamenetek: `src/app/api/sessions' (GET) -- Díjkorlátok: "src/app/api/rate-limits" (GET) -- Rugalmasság: "src/app/api/resilience" (GET/PATCH) – szolgáltatói profilok, megszakító, sebességkorlát állapot -- Rugalmasság visszaállítása: `src/app/api/resilience/reset' (POST) – megszakítók visszaállítása + lehűlés -- Gyorsítótár statisztikái: "src/app/api/cache/stats" (GET/DELETE) -- A modell elérhetősége: "src/app/api/models/availability" (GET/POST) -- Telemetria: "src/app/api/telemetry/summary" (GET) -- Költségkeret: `src/app/api/usage/budget' (GET/POST) -- Tartalék láncok: `src/app/api/fallback/chains' (GET/POST/DELETE) -- Megfelelőségi ellenőrzés: `src/app/api/compliance/audit-log' (GET) -- Evals: "src/app/api/evals" (GET/POST), "src/app/api/evals/[suiteId]" (GET) -- Irányelvek: `src/app/api/policies' (GET/POST)## 2) SSE + Translation Core +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -Fő áramlási modulok: +## 2) SSE + Translation Core -- Bejegyzés: `src/sse/handlers/chat.ts` -- Alapvető hangszerelés: "open-sse/handlers/chatCore.ts" -- Szolgáltató végrehajtási adapterei: `open-sse/executors/*` -- Formátumészlelés/szolgáltató konfigurációja: "open-sse/services/provider.ts" -- Modell parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- A fiók tartalék logikája: `open-sse/services/accountFallback.ts` -- Fordítási nyilvántartás: "open-sse/translator/index.ts". -- Adatfolyam-átalakítások: "open-sse/utils/stream.ts", "open-sse/utils/streamHandler.ts" -- Használat kibontása/normalizálása: "open-sse/utils/usageTracking.ts" +Main flow modules: + +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` - Think tag parser: `open-sse/utils/thinkTagParser.ts` -- Beágyazáskezelő: `open-sse/handlers/embeddings.ts` -- Beágyazási szolgáltató nyilvántartása: "open-sse/config/embeddingRegistry.ts" -- Képgeneráló kezelő: `open-sse/handlers/imageGeneration.ts` -- Képszolgáltató regisztrációs adatbázisa: `open-sse/config/imageRegistry.ts` -- A válaszok fertőtlenítése: "open-sse/handlers/responseSanitizer.ts" -- Szerepkör normalizálása: `open-sse/services/roleNormalizer.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -Szolgáltatások (üzleti logika): +Services (business logic): -- Fiókválasztás/pontozás: "open-sse/services/accountSelector.ts" -- Kontextus-életciklus-kezelés: `open-sse/services/contextManager.ts` -- IP-szűrő végrehajtása: `open-sse/services/ipFilter.ts` -- Munkamenetkövetés: `open-sse/services/sessionManager.ts` -- Deduplikáció kérése: "open-sse/services/signatureCache.ts" -- Rendszerprompt injekció: `open-sse/services/systemPrompt.ts` -- Gondolkodó költségvetés-kezelés: `open-sse/services/thinkingBudget.ts` -- Helyettesítő karakteres modell-útválasztás: "open-sse/services/wildcardRouter.ts" -- Díjkorlát kezelése: `open-sse/services/rateLimitManager.ts` -- Megszakító: "open-sse/services/circuitBreaker.ts" +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -Domain réteg modulok: +Domain layer modules: -- A modell elérhetősége: `src/lib/domain/modelAvailability.ts` -- Költségszabályok/költségkeretek: `src/lib/domain/costRules.ts` -- Tartalék házirend: `src/lib/domain/fallbackPolicy.ts` -- Kombinált feloldó: `src/lib/domain/comboResolver.ts` -- Kizárási szabályzat: `src/lib/domain/lockoutPolicy.ts` -- Házirend motor: `src/domain/policyEngine.ts` — központosított zárolás → költségvetés → tartalék kiértékelés -- Hibakód-katalógus: "src/lib/domain/errorCodes.ts". -- Kérelemazonosító: `src/lib/domain/requestId.ts` -- Lekérési időtúllépés: `src/lib/domain/fetchTimeout.ts` -- Telemetria kérése: `src/lib/domain/requestTelemetry.ts` -- Megfelelőség/ellenőrzés: `src/lib/domain/compliance/index.ts` +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` - Eval runner: `src/lib/domain/evalRunner.ts` -- Tartomány állapotának fennmaradása: `src/lib/db/domainState.ts' — SQLite CRUD tartalék láncokhoz, költségvetésekhez, költségelőzményekhez, zárolási állapothoz, megszakítókhoz +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -OAuth-szolgáltató modulok (12 külön fájl az `src/lib/oauth/providers/` alatt): +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): - Registry index: `src/lib/oauth/providers/index.ts` -- Egyéni szolgáltatók: "claude.ts", "codex.ts", "gemini.ts", "antigravity.ts", "qoder.ts", "qwen.ts", "kimi-coding.ts", "github.ts", "kiro.ts", "codex". `cline.ts` -- Vékony burkolóanyag: `src/lib/oauth/providers.ts' – újraexportálás az egyes modulokból## 3) Persistence Layer +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -Elsődleges állapotú DB (SQLite): +## 3) Persistence Layer -- Alapvető infrastruktúra: `src/lib/db/core.ts` (better-sqlite3, migrációk, WAL) -- Homlokzat újraexportálása: "src/lib/localDb.ts" (vékony kompatibilitási réteg a hívók számára) -- fájl: `${DATA_DIR}/storage.sqlite` (vagy `$XDG_CONFIG_HOME/omniroute/storage.sqlite`, ha be van állítva, különben `~/.omniroute/storage.sqlite`) -- entitások (táblák + KV névterek): szolgáltatói kapcsolatok, szolgáltató csomópontok, modellálnevek, kombók, apiKeys, beállítások, árképzés,**customModels**,**proxyConfig**,**ipFilter**,**thhinkingBudget**,**systemPrompt** +Primary state DB (SQLite): -Használat tartóssága: +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -- homlokzat: `src/lib/usageDb.ts` (bontott modulok az `src/lib/usage/*` fájlban) -- SQLite táblák a "storage.sqlite" fájlban: "használati_előzmények", "hívásnaplók", "proxy_naplók" -- az opcionális melléktermékek a kompatibilitáshoz/hibakereséshez maradnak (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- A régebbi JSON-fájlokat a rendszer indítási migrációval költözteti az SQLite-ba, ha vannak +Usage persistence: + +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present Domain State DB (SQLite): -- "src/lib/db/domainState.ts" - CRUD műveletek a tartomány állapotához -- Táblázatok (az "src/lib/db/core.ts" fájlban létrehozva): "domain_fallback_chains", "domain_budgets", "domain_cost_history", "domain_lockout_state", "domain_circuit_breakers" -- Átírási gyorsítótár minta: a memórián belüli térképek mérvadóak futás közben; a mutációk szinkronban íródnak az SQLite-ba; állapot visszaáll a DB-ből hidegindításkor## 4) Auth + Security Surfaces +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start -- Az irányítópult cookie hitelesítése: "src/proxy.ts", "src/app/api/auth/login/route.ts" -- API-kulcs létrehozása/ellenőrzése: `src/shared/utils/apiKey.ts` -- A szolgáltatói titkok megmaradtak a "providerConnections" bejegyzésekben -- Kimenő proxy támogatása az "open-sse/utils/proxyFetch.ts" (env vars) és az "open-sse/utils/networkProxy.ts" segítségével (szolgáltatónként vagy globálisan konfigurálható)## 5) Cloud Sync +## 4) Auth + Security Surfaces -- Ütemező init: "src/lib/initCloudSync.ts", "src/shared/services/initializeCloudSync.ts", "src/shared/services/modelSyncScheduler.ts" -- Időszakos feladat: `src/shared/services/cloudSyncScheduler.ts` -- Időszakos feladat: `src/shared/services/modelSyncScheduler.ts` -- Útvonal vezérlése: "src/app/api/sync/cloud/route.ts"## Request Lifecycle (`/v1/chat/completions`) +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -A tartalék döntéseket az `open-sse/services/accountFallback.ts` vezérli állapotkódok és hibaüzenet-heurisztika használatával. A kombinált útválasztás egy plusz védelmet ad: a szolgáltatói hatókörű 400-asokat, mint például az upstream tartalomblokkolás és a szerepérvényesítési hibák, modellhelyi hibákként kezelik, így a későbbi kombinált célok továbbra is futhatnak.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -Az élő forgalom alatti frissítés az `open-sse/handlers/chatCore.ts` fájlban, a `refreshCredentials()` végrehajtón keresztül történik.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -Az időszakos szinkronizálást a „CloudSyncScheduler” indítja el, ha a felhő engedélyezve van.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -Fizikai tároló fájlok: +Physical storage files: -- elsődleges futásidejű DB: `${DATA_DIR}/storage.sqlite` -- kérésnapló sorai: `${DATA_DIR}/log.txt` (kompat/debug melléktermék) -- Strukturált hívások hasznosadat-archívuma: `${DATA_DIR}/call_logs/` -- opcionális fordítói/hibakereső munkamenetek kérése: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: kompatibilitási API-k -- `src/app/api/v1/providers/[szolgáltató]/\*: szolgáltatónként dedikált útvonalak (csevegés, beágyazások, képek) -- `src/app/api/providers*`: szolgáltató CRUD, érvényesítés, tesztelés -- `src/app/api/provider-nodes*`: egyéni kompatibilis csomópontkezelés -- "src/app/api/provider-models": egyéni modellkezelés (CRUD) -- "src/app/api/models/route.ts": modellkatalógus API (álnevek + egyéni modellek) -- `src/app/api/oauth/*`: OAuth/eszközkód folyamatok -- `src/app/api/keys*`: helyi API kulcs életciklusa -- `src/app/api/models/alias`: alias kezelése -- `src/app/api/combos*`: tartalék kombinált kezelés -- "src/app/api/pricing": az árképzés felülbírálása a költségszámításhoz -- "src/app/api/settings/proxy": proxykonfiguráció (GET/PUT/DELETE) -- "src/app/api/settings/proxy/test": kimenő proxykapcsolati teszt (POST) -- `src/app/api/usage/*`: használati és naplózási API-k -- `src/app/api/sync/*` + `src/app/api/cloud/*`: felhőszinkronizálás és felhő felé néző segítők -- `src/app/api/cli-tools/*`: helyi CLI konfigurációs írók/ellenőrzők -- "src/app/api/settings/ip-filter": IP engedélyezési lista/blokkolista (GET/PUT) -- "src/app/api/settings/thhinking-budget": gondolkodási jogkivonat költségvetési konfigurációja (GET/PUT) -- "src/app/api/settings/system-prompt": globális rendszerprompt (GET/PUT) -- "src/app/api/sessions": aktív munkamenet-lista (GET) -- "src/app/api/rate-limits": fiókonkénti kamatkorlát állapota (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: kéréselemzés, kombinált kezelés, fiókválasztó hurok -- "open-sse/handlers/chatCore.ts": fordítás, végrehajtó feladás, újrapróbálkozás/frissítés kezelése, adatfolyam beállítása -- `open-sse/executors/*`: szolgáltató-specifikus hálózat- és formátumviselkedés### Translation Registry and Format Converters +### Routing and Execution Core -- "open-sse/translator/index.ts": fordítói nyilvántartás és hangszerelés -- Fordítók kérése: `open-sse/translator/request/*` -- Válasz fordítók: `open-sse/translator/response/*` -- Formátumkonstansok: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: állandó konfiguráció/állapot és tartomány fennmaradása az SQLite-on -- `src/lib/localDb.ts`: DB modulok kompatibilitási újraexportálása -- `src/lib/usageDb.ts`: a használati előzmények/hívásnaplók homlokzata az SQLite táblák tetején## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Minden szolgáltató rendelkezik egy speciális végrehajtóval, amely kiterjeszti a „BaseExecutort” (az „open-sse/executors/base.ts” fájlban), amely URL-építést, fejléc-építést, újrapróbálkozást exponenciális visszalépéssel, hitelesítő adatok frissítését és az „execute()” hangszerelési metódust biztosítja. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Végrehajtó | Szolgáltató(k) | Különleges kezelés | -| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------------- | -| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dinamikus URL/fejléc konfiguráció szolgáltatónként | -| "AntigravityExecutor" | Google Antigravitáció | Egyéni projekt/munkamenet azonosítók, Újrapróbálkozás-elemzés után | -| "CodexExecutor" | OpenAI Codex | Rendszerutasításokat szúr be, érvelési erőfeszítést kényszerít | -| "CursorExecutor" | Kurzor IDE | ConnectRPC protokoll, Protobuf kódolás, kérés aláírása ellenőrző összeggel | -| "GithubExecutor" | GitHub másodpilóta | Másodpilóta token frissítése, VSCode-utánzó fejlécek | -| "KiroExecutor" | AWS CodeWhisperer/Kiro | AWS EventStream bináris formátum → SSE konverzió | -| "GeminiCLIExecutor" | Gemini CLI | Google OAuth-token frissítési ciklus | +### Persistence -Minden más szolgáltató (beleértve az egyéni kompatibilis csomópontokat is) a "DefaultExecutor"-t használja.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Szolgáltató | Formátum | Auth | Stream | Nem adatfolyam | Token Refresh | Használati API | -| ------------------ | ---------------- | ------------------------- | ------------------ | -------------- | ------------- | ------------------------- | ------------------------------ | -| Claude | claude | API kulcs / OAuth | ✅ | ✅ | ✅ | ⚠️ Csak adminisztrátor | -| Ikrek | ikrek | API kulcs / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | -| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | -| Antigravitáció | antigravitáció | OAuth | ✅ | ✅ | ✅ | ✅ Teljes kvóta API | -| OpenAI | openai | API kulcs | ✅ | ✅ | ❌ | ❌ | -| Codex | openai-responses | OAuth | ✅ kényszer | ❌ | ✅ | ✅ Díjkorlátok | -| GitHub másodpilóta | openai | OAuth + másodpilóta token | ✅ | ✅ | ✅ | ✅ Kvóta pillanatképek | -| Kurzor | kurzor | Egyéni ellenőrző összeg | ✅ | ✅ | ❌ | ❌ | -| Kiro | kiro | AWS SSO OIDC | ✅ (Eseményfolyam) | ❌ | ✅ | ✅ Felhasználási korlátok | -| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Kérésre | -| Qoder | openai | OAuth (alap) | ✅ | ✅ | ✅ | ⚠️ Kérésre | -| OpenRouter | openai | API kulcs | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | claude | API kulcs | ✅ | ✅ | ❌ | ❌ | -| DeepSeek | openai | API kulcs | ✅ | ✅ | ❌ | ❌ | -| Groq | openai | API kulcs | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | openai | API kulcs | ✅ | ✅ | ❌ | ❌ | -| Mistral | openai | API kulcs | ✅ | ✅ | ❌ | ❌ | -| Zavartság | openai | API kulcs | ✅ | ✅ | ❌ | ❌ | -| Együtt AI | openai | API kulcs | ✅ | ✅ | ❌ | ❌ | -| Tűzijáték AI | openai | API kulcs | ✅ | ✅ | ❌ | ❌ | -| Cerebrák | openai | API kulcs | ✅ | ✅ | ❌ | ❌ | -| Cohere | openai | API kulcs | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | openai | API kulcs | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Az észlelt forrásformátumok a következők: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. -- "openai". -- "Openai-responses". +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | + +All other providers (including custom compatible nodes) use the `DefaultExecutor`. + +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: + +- `openai` +- `openai-responses` - `claude` -- "ikrek". +- `gemini` -A célformátumok a következők: +Target formats include: -- OpenAI chat/válaszok +- OpenAI chat/Responses - Claude -- Gemini/Gemini-CLI/Antigravitációs boríték +- Gemini/Gemini-CLI/Antigravity envelope - Kiro -- Kurzor +- Cursor -A fordítások az**OpenAI-t használják hub-formátumként**– minden konverzió köztesként az OpenAI-n megy keresztül:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -A fordítások kiválasztása dinamikusan történik a forrás hasznos adat alakja és a szolgáltató célformátuma alapján. +Additional processing layers in the translation pipeline: -További feldolgozási rétegek a fordítási folyamatban: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Választisztítás**– Megszünteti a nem szabványos mezőket az OpenAI-formátumú válaszoktól (mind az adatfolyam-, mind a nem adatfolyam-küldéstől) a szigorú SDK-megfelelőség biztosítása érdekében --**Szerepkör normalizálása**— Átalakítja a "fejlesztő" → "rendszert" nem OpenAI-célokhoz; egyesíti a "rendszer" → "felhasználó" paramétert a rendszerszerepkört elutasító modellekhez (GLM, ERNIE) --**Think címke kivonatolás**– A tartalomból a `...` blokkokat elemzi a `reasoning_content` mezőbe --**Strukturált kimenet**- Az OpenAI `response_format.json_schema`-t a Gemini `responseMimeType` + `responseSchema`-jává alakítja## Supported API Endpoints +## Supported API Endpoints -| Végpont | Formátum | Kezelő | -| --------------------------------------------------- | ------------------- | -------------------------------------------------------------------- | -| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | -| `POST /v1/messages` | Claude Üzenetek | Ugyanaz a kezelő (automatikusan észlelve) | -| `POST /v1/responses` | OpenAI válaszok | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/embeddings` | OpenAI beágyazások | `open-sse/handlers/embeddings.ts` | -| `GET /v1/embeddings` | Modell lista | API útvonal | -| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | -| `GET /v1/images/generations` | Modell lista | API útvonal | -| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedikált szolgáltatónként modellellenőrzéssel | -| `POST /v1/providers/{provider}/embeddings` | OpenAI beágyazások | Dedikált szolgáltatónként modellellenőrzéssel | -| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedikált szolgáltatónként modellellenőrzéssel | -| `POST /v1/messages/count_tokens` | Claude Token Count | API útvonal | -| `GET /v1/models` | OpenAI modellek listája | API útvonal (csevegés + beágyazás + kép + egyéni modellek) | -| `GET /api/models/catalog` | Katalógus | Minden modell szolgáltató + típus szerint csoportosítva | -| `POST /v1beta/models/*:streamGenerateContent` | Ikrek bennszülött | API route | -| `GET/PUT/DELETE /api/settings/proxy` | Proxy konfiguráció | Hálózati proxy konfiguráció | -| `POST /api/settings/proxy/test` | Proxy kapcsolat | Proxy állapot/kapcsolati teszt végpontja | -| `GET/POST/DELETE /api/provider-models` | Szolgáltatói modellek | A szolgáltatói modell metaadatainak háttere egyéni és felügyelt elérhető modellek |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -A bypass kezelő (`open-sse/utils/bypassHandler.ts`) elfogja a Claude CLI ismert "kidobási" kéréseit – bemelegítő pingeket, címkivonatokat és tokenszámlálásokat –, és**hamis választ**ad vissza anélkül, hogy felemészti a szolgáltatói tokeneket. Ez csak akkor aktiválódik, ha a „User-Agent” tartalmazza a „claude-cli”-t.## Request Logger Pipeline +## Bypass Handler -A kérésnaplózó (`open-sse/utils/requestLogger.ts`) egy 7 szakaszból álló hibakeresési naplózási folyamatot biztosít, amely alapértelmezés szerint le van tiltva, és az `ENABLE_REQUEST_LOGS=true` paraméterrel engedélyezve van:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -A fájlok a `/logs//` mappába íródnak minden egyes kérési munkamenethez.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- szolgáltatói fiók lehűtése tranziens/sebesség/hitelesítési hibák esetén -- tartalék fiók a sikertelen kérés előtt -- kombinált modell tartalék, ha az aktuális modell/szolgáltató elérési útja kimerült## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- Előzetes ellenőrzés és frissítés újrapróbálkozással a frissíthető szolgáltatóknál -- 401/403 újrapróbálkozás frissítési kísérlet után az alapútvonalon## 3) Stream Safety +## 2) Token Expiry -- leválasztást érzékelő streamvezérlő -- fordítási adatfolyam a folyamvégi öblítéssel és `[KÉSZ]` kezeléssel -- a használati becslés tartaléka, ha hiányoznak a szolgáltató használati metaadatai## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- szinkronizálási hibák jelennek meg, de a helyi futásidő folytatódik -- Az ütemező rendelkezik újrapróbálkozásra alkalmas logikával, de az időszakos végrehajtás jelenleg alapértelmezés szerint egykísérletű szinkronizálást hív meg## 5) Data Integrity +## 3) Stream Safety + +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing + +## 4) Cloud Sync Degradation + +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default + +## 5) Data Integrity - SQLite schema migrations and auto-upgrade hooks at startup -- örökölt JSON → SQLite migrációs kompatibilitási útvonal## Observability and Operational Signals +- legacy JSON → SQLite migration compatibility path + +## Observability and Operational Signals Runtime visibility sources: -- konzolnaplók innen: `src/sse/utils/logger.ts` -- kérésenkénti használati aggregátumok az SQLite-ban (`usage_history`, `call_logs`, `proxy_logs`) -- négylépcsős részletes hasznos adatrögzítés az SQLite-ben (`request_detail_logs`), amikor a `settings.detailed_logs_enabled=true` -- szöveges kérés állapotnaplózása a `log.txt' fájlban (opcionális/kompat) -- opcionális mélykérési/fordítási naplók a "logs/" alatt, ha "ENABLE_REQUEST_LOGS=true" -- irányítópult-használati végpontok (`/api/usage/*`) a felhasználói felület felhasználásához +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -A részletes kérés hasznos adatrögzítése irányított hívásonként legfeljebb négy JSON-adatterhelési szakaszt tárol: +Detailed request payload capture stores up to four JSON payload stages per routed call: -- nyers kérés érkezett az ügyféltől -- a lefordított kérés ténylegesen felfelé küldve -- a szolgáltatói válasz JSON-ként rekonstruálva; a streamelt válaszokat a végső összefoglalóba és az adatfolyam metaadataiba tömörítik -- az OmniRoute által visszaküldött végső ügyfélválasz; a streamelt válaszokat ugyanabban a tömör összefoglaló formában tároljuk## Security-Sensitive Boundaries +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form -- A JWT titkos (`JWT_SECRET`) biztosítja az irányítópult munkamenet-cookie-ellenőrzését/aláírását -- A kezdeti jelszó-betöltést (`INITIAL_PASSWORD`) kifejezetten be kell állítani az első futtatáshoz -- API kulcs HMAC titkos (`API_KEY_SECRET`) biztosítja a generált helyi API kulcs formátumot -- A szolgáltatói titkok (API-kulcsok/tokenek) megmaradnak a helyi adatbázisban, és fájlrendszer-szinten védeni kell őket -- A felhőszinkronizálási végpontok API kulcs hitelesítés + gépazonosító szemantikára támaszkodnak## Environment and Runtime Matrix +## Security-Sensitive Boundaries -A kód által aktívan használt környezeti változók: +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics -- Alkalmazás/hitelesítés: "JWT_SECRET", "INITIAL_PASSWORD" -- Tárhely: `DATA_DIR` -- Kompatibilis csomópont viselkedése: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Opcionális tárhely-alap-felülírás (Linux/macOS, ha a "DATA_DIR" nincs beállítva): "XDG_CONFIG_HOME" -- Biztonsági kivonat: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Naplózás: `ENABLE_REQUEST_LOGS` -- Szinkronizálás/felhő URL-elés: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- Kimenő proxy: "HTTP_PROXY", "HTTPS_PROXY", "ALL_PROXY", "NO_PROXY" és kisbetűs változatai -- SOCKS5 funkciójelzők: "ENABLE_SOCKS5_PROXY", "NEXT_PUBLIC_ENABLE_SOCKS5_PROXY" -- Platform/futásidejű segítők (nem alkalmazás-specifikus konfiguráció): "APPDATA", "NODE_ENV", "PORT", "HOSTNAME"## Known Architectural Notes +## Environment and Runtime Matrix -1. A `usageDb` és a `localDb` ugyanazon az alapkönyvtár-házirenden (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) osztozik a régi fájlmigrációval. -2. Az `/api/v1/route.ts' ugyanahhoz az egyesített katalóguskészítőhöz delegálódik, amelyet a `/api/v1/models' (`src/app/api/v1/models/catalog.ts`) használ a szemantikai eltolódás elkerülése érdekében. -3. A kérésnaplózó teljes fejlécet/törzsöt ír, ha engedélyezve van; a naplókönyvtárat érzékenyként kezeli. -4. A felhő viselkedése a helyes "NEXT_PUBLIC_BASE_URL" és a felhő-végpont elérhetőségétől függ. -5. Az `open-sse/` könyvtár `@omniroute/open-sse`**npm munkaterület-csomagként**kerül közzétételre. A forráskód az `@omniroute/open-sse/...` (a Next.js `transpilePackages`) segítségével importálja. A dokumentum elérési útjai továbbra is az `open-sse/` könyvtárnevet használják a következetesség érdekében. -6. Az irányítópulton lévő diagramok**Újragrafikonokat**(SVG-alapú) használnak az elérhető, interaktív analitikai vizualizációkhoz (modellhasználati sávdiagramok, szolgáltatói bontási táblázatok sikerarányokkal). -7. Az E2E-tesztek a**Playwright**-ot használják (`tests/e2e/`), az `npm run test:e2e`-n keresztül futnak. Az egységtesztek a**Node.js tesztfutót**(`tests/unit/`) használják, az `npm run test:unit` segítségével futnak. Az `src/` alatti forráskód**TypeScript**(`.ts`/`.tsx`); az `open-sse/` munkaterület továbbra is JavaScript (`.js`). -8. A Beállítások oldal 5 lapra van felosztva: Biztonság, Útválasztás (6 globális stratégia: kitöltés-első, kör-robin, p2c, véletlenszerű, legkevésbé használt, költségoptimalizált), Rugalmasság (szerkeszthető sebességkorlátok, megszakító, házirendek), AI (gondolkodó költségvetés, rendszerkérdés, gyorsítótár), Speciális (proxy).## Operational Verification Checklist +Environment variables actively used by code: -- Build forrásból: `npm run build` -- Build Docker kép: `docker build -t omniroute .` -- Indítsa el a szervizt és ellenőrizze: +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: - `GET /api/settings` - `GET /api/v1/models` -- A CLI cél alap URL-jének `http://:20128/v1` kell lennie, ha `PORT=20128` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/hu/docs/FEATURES.md b/docs/i18n/hu/docs/FEATURES.md index b255f56343..0d7380ac06 100644 --- a/docs/i18n/hu/docs/FEATURES.md +++ b/docs/i18n/hu/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Vizuális útmutató az OmniRoute irányítópult minden részéhez.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -AI-szolgáltatói kapcsolatok kezelése: OAuth-szolgáltatók (Claude Code, Codex, Gemini CLI), API-kulcs-szolgáltatók (Groq, DeepSeek, OpenRouter) és ingyenes szolgáltatók (Qoder, Qwen, Kiro). A Kiro-számlák tartalmazzák a hitelegyenleg nyomon követését – a fennmaradó kreditek, a teljes juttatás és a megújítási dátum látható az Irányítópult → Használat menüpontban.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Hozzon létre modell-útválasztási kombókat 6 stratégiával: prioritás, súlyozott, kör-robin, véletlenszerű, legkevésbé használt és költségoptimalizált. Mindegyik kombó több modellt láncol össze automatikus visszaállítással, valamint gyors sablonokat és készenléti ellenőrzéseket tartalmaz.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Átfogó használati elemzés token-fogyasztással, költségbecslésekkel, tevékenységi hőtérképekkel, heti elosztási diagramokkal és szolgáltatónkénti lebontásokkal.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Valós idejű megfigyelés: üzemidő, memória, verzió, késleltetési százalékok (p50/p95/p99), gyorsítótár-statisztika és szolgáltatói megszakító állapotok.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Négy mód az API-fordítások hibakeresésére:**Playground**(formátum-átalakító),**Chat Tester**(élő kérések),**Test Bench**(kötegelt tesztek) és**Élő figyelő**(valós idejű adatfolyam).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Teszteljen bármely modellt közvetlenül a műszerfalról. Válassza ki a szolgáltatót, a modellt és a végpontot, írjon promptokat a Monaco Editor segítségével, streamelje a válaszokat valós időben, szakítsa meg a streamelést, és tekintse meg az időzítési mutatókat.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Testreszabható színtémák a teljes műszerfalhoz. Válasszon a 7 előre beállított szín közül (korall, kék, piros, zöld, ibolya, narancs, cián), vagy hozzon létre egyéni témát bármilyen hatszögletű szín kiválasztásával. Támogatja a világos, sötét és rendszermódot.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Átfogó beállítási panel fülekkel: +Comprehensive settings panel with tabs: --**Általános**- Rendszertárolás, biztonsági mentések kezelése (export/import adatbázis) -**Megjelenés**- Témaválasztó (sötét/világos/rendszer), színtéma előre beállított és egyéni színek, állapotnapló láthatósága, oldalsáv elemláthatósági vezérlői -**Biztonság**– API-végpont védelem, egyéni szolgáltatói blokkolás, IP-szűrés, munkamenet-információk -**Útválasztás**— Modellálnevek, háttérfeladat-romlás -**Rugalmasság**- Díjkorlátok fennmaradása, megszakító hangolása, letiltott fiókok automatikus letiltása, szolgáltató lejáratának figyelése -**Speciális**— Konfiguráció felülbírálása, konfigurációs ellenőrzési nyomvonal, tartalék rontási mód![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Egy kattintással konfigurálható mesterséges intelligencia kódoló eszközök: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor és Factory Droid. Automatikus konfigurációs alkalmazás/visszaállítás, csatlakozási profilok és modellleképezés.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Irányítópult a CLI-ügynökök felfedezéséhez és kezeléséhez. 14 beépített ügynökből álló rácsot jelenít meg (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) a következőkkel: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Telepítés állapota**- Telepítve / Nem található verzióérzékeléssel -**Protokoll jelvények**— stdio, HTTP stb. -**Egyéni ügynökök**— Regisztráljon bármilyen CLI-eszközt űrlapon keresztül (név, bináris, verzióparancs, spawn args) -**CLI ujjlenyomat-egyeztetés**– szolgáltatónkénti váltás a natív CLI-kérés aláírásainak egyeztetésére, csökkentve a kitiltási kockázatot, miközben megőrzi a proxy IP-címét--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Hozzon létre képeket, videókat és zenét az irányítópultról. Támogatja az OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open és MusicGen alkalmazásokat.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Valós idejű kérések naplózása szolgáltató, modell, fiók és API kulcs szerinti szűréssel. Megjeleníti az állapotkódokat, a tokenhasználatot, a várakozási időt és a válasz részleteit.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Az Ön egységes API-végpontja a képességek lebontásával: csevegés befejezése, válaszok API, beágyazások, képgenerálás, átsorolás, hangátírás, szövegfelolvasó, moderálás és regisztrált API-kulcsok. Cloudflare Quick Tunnel integráció és felhőproxy támogatás a távoli eléréshez.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -API-kulcsok létrehozása, hatóköre és visszavonása. Minden kulcs korlátozható meghatározott modellekre/szolgáltatókra teljes hozzáféréssel vagy csak olvasási engedéllyel. Vizuális kulcskezelés a használat nyomon követésével.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Adminisztratív műveletek követése művelettípus, szereplő, cél, IP-cím és időbélyeg szerinti szűréssel. Teljes biztonsági eseménytörténet.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Native Electron asztali alkalmazás Windows, macOS és Linux rendszerhez. Futtassa az OmniRoute-ot önálló alkalmazásként rendszertálca-integrációval, offline támogatással, automatikus frissítéssel és egy kattintással történő telepítéssel. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Főbb jellemzők: +Key features: -- Szerver készenléti lekérdezés (hidegindításkor nincs üres képernyő) -- Rendszertálca portkezeléssel -- Tartalombiztonsági szabályzat -- Egypéldányos zár -- Automatikus frissítés újraindításkor -- Platform-feltételes felhasználói felület (macOS közlekedési lámpák, Windows/Linux alapértelmezett címsor) -- Megerősített Electron összeállítási csomagolás – az önálló csomagban lévő szimbolizált "csomópont_modulok" felismerése és elutasítása a csomagolás előtt, megakadályozva a futásidejű függőséget az összeállítási géptől (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Lásd: [`electron/README.md`](../electron/README.md) a teljes dokumentációért. +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/hu/docs/TROUBLESHOOTING.md b/docs/i18n/hu/docs/TROUBLESHOOTING.md index ee601d7e84..6719ac515a 100644 --- a/docs/i18n/hu/docs/TROUBLESHOOTING.md +++ b/docs/i18n/hu/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -Az OmniRoute gyakori problémái és megoldásai.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Probléma | Megoldás | -| ----------------------------------- | ----------------------------------------------------------------------------------------------------- | --- | -| Az első bejelentkezés nem működik | Állítsa be az 'INITIAL_PASSWORD' értéket a '.env'-ben (nincs merevkódolt alapértelmezés) | -| A műszerfal rossz porton nyílik meg | Állítsa be: `PORT=20128` és `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Nincs kérésnapló a `logs/` | alatt Állítsa be a következőt: `ENABLE_REQUEST_LOGS=true' | -| EACCES: engedély megtagadva | A `~/.omniroute` felülbírálásához állítsa be a `DATA_DIR=/útvonal/útvonal/írható/könyvtár` paramétert | -| Az útválasztási stratégia nem menti | Frissítés v1.4.11+ verzióra (Zod-séma javítása a beállítások fennmaradásához) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Ok:**A szolgáltatói kvóta kimerült. +**Cause:** Provider quota exhausted. -**Javítás:** +**Fix:** -1. Ellenőrizze az irányítópult kvótakövetőjét -2. Használjon kombót tartalék szintekkel -3. Váltson olcsóbb/ingyenes szintre### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Ok:**Az előfizetési kvóta kimerült. +### Rate Limiting -**Javítás:** +**Cause:** Subscription quota exhausted. -- Tartalék hozzáadása: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Használja a GLM/MiniMax-ot olcsó tartalékként### OAuth Token Expired +**Fix:** -Az OmniRoute automatikusan frissíti a tokeneket. Ha a problémák továbbra is fennállnak: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Irányítópult → Szolgáltató → Újracsatlakozás -2. Törölje és adja hozzá újra a szolgáltatói kapcsolatot--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Ellenőrizze, hogy a „BASE_URL” a futó példányra mutat (pl. „http://localhost:20128”). -2. Ellenőrizze, hogy a „CLOUD_URL” a felhő végpontjára mutat (pl. „https://omniroute.dev”). -3. Tartsa a `NEXT_PUBLIC_*` értékeket a szerveroldali értékekkel összhangban### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Tünet:**`Váratlan 'd'... token a felhő-végponton nem streaming hívásokhoz. +### Cloud `stream=false` Returns 500 -**Ok:**Az Upstream SSE hasznos adatot ad vissza, miközben az ügyfél a JSON-t várja. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Megkerülő megoldás:**Használja a „stream=true” paramétert a felhőalapú közvetlen hívásokhoz. A helyi futási környezet tartalmazza az SSE→JSON tartalékot.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Hozzon létre egy új kulcsot a helyi irányítópultról (`/api/keys`) -2. Futtassa a felhőszinkronizálást: Engedélyezze a Felhőt → Szinkronizálás most -3. A régi/nem szinkronizált kulcsok továbbra is „401”-et adhatnak vissza felhőben--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Ellenőrizze a futásidejű mezőket: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. Hordozható módban: használja a "runner-cli" képcélzást (csomagolt CLI-k) -3. Gazda beillesztési módhoz: állítsa be a `CLI_EXTRA_PATHS'-t és a mount bin könyvtárat csak olvashatóként -4. Ha `installed=true` és `runnable=false`: bináris fájl található, de az állapotellenőrzés sikertelen### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Ellenőrizze a használati statisztikákat az Irányítópult → Használat menüpontban -2. Állítsa át az elsődleges modellt GLM/MiniMax-ra -3. Használjon ingyenes réteget (Gemini CLI, Qoder) a nem kritikus feladatokhoz -4. Állítsa be a költségkereteket API-kulcsonként: Irányítópult → API-kulcsok → Költségvetés--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Állítsa be az „ENABLE_REQUEST_LOGS=true” értéket az „.env” fájlban. A naplók a `logs/` könyvtárban jelennek meg.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,103 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Fő állapot: `${DATA_DIR}/storage.sqlite` (szolgáltatók, kombinációk, álnevek, kulcsok, beállítások) -- Használat: SQLite táblák a `storage.sqlite` fájlban (`használati_előzmények`, `hívásnaplók`, `proxy_logs`) + opcionális `${DATA_DIR}/log.txt` és `${DATA_DIR}/call_logs/` -- Naplók lekérése: `/logs/...` (amikor `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Amikor egy szolgáltató megszakítója NYITVA van, a kérések blokkolva vannak, amíg a leállás le nem jár. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Javítás:** +**Fix:** -1. Lépjen az**Irányítópult → Beállítások → Rugalmasság**menüpontra. -2. Ellenőrizze az érintett szolgáltató megszakítókártyáját -3. Kattintson a**Reset All**elemre az összes megszakító törléséhez, vagy várja meg, amíg a lehűlés lejár -4. A visszaállítás előtt ellenőrizze, hogy a szolgáltató valóban elérhető-e### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Ha egy szolgáltató ismételten NYITOTT állapotba lép: +### Provider keeps tripping the circuit breaker -1. Ellenőrizze a**Irányítópult → Állapot → Szolgáltató állapota**menüpontban a hibamintát -2. Lépjen a**Beállítások → Ellenállás → Szolgáltatói profilok**menüpontra, és növelje a meghibásodási küszöböt. -3. Ellenőrizze, hogy a szolgáltató megváltoztatta-e az API-korlátokat, vagy nem igényel-e újbóli hitelesítést -4. Tekintse át a késleltetési telemetriát – a magas késleltetés időtúllépésen alapuló hibákat okozhat--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Győződjön meg arról, hogy a megfelelő előtagot használja: "deepgram/nova-3" vagy "assemblyai/best" -- Ellenőrizze, hogy a szolgáltató csatlakoztatva van-e az**Irányítópult → Szolgáltatók**menüpontban.### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Ellenőrizze a támogatott hangformátumokat: "mp3", "wav", "m4a", "flac", "ogg", "webm" -- Ellenőrizze, hogy a fájl mérete a szolgáltatói korlátokon belül van (általában < 25 MB) -- Ellenőrizze a szolgáltatói API kulcs érvényességét a szolgáltatói kártyán--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Használja az**Irányítópult → Fordító**lehetőséget a formátumfordítási problémák elhárításához: +Use **Dashboard → Translator** to debug format translation issues: -| mód | Mikor kell használni | -| --------------------- | ----------------------------------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Játszótér** | Hasonlítsa össze a bemeneti/kimeneti formátumokat egymás mellett – illesszen be egy hibás kérést, hogy megtudja, hogyan fordítja le | -| **Csevegés tesztelő** | Küldjön élő üzeneteket, és ellenőrizze a teljes kérés/válasz hasznos adatot, beleértve a fejléceket | -| **Próbapad** | Futtasson kötegelt teszteket a formátumkombinációk között, hogy megtudja, mely fordítások hibásak | -| **Élő monitor** | Nézze meg a valós idejű kérések folyamatát az időszakos fordítási problémák észleléséhez | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Nem jelennek meg a gondolkodási címkék**— Ellenőrizze, hogy a célszolgáltató támogatja-e a gondolkodást és a gondolkodási költségvetés beállítását -**Eszközhívások megszakítása**— Egyes formátumfordítások eltávolíthatják a nem támogatott mezőket; ellenőrizze Playground módban -**Rendszerprompt hiányzik**— Claude és Gemini fogantyúrendszere eltérő módon szól; ellenőrizze a fordítás kimenetét -**Az SDK nyers karakterláncot ad vissza az objektum helyett**- Javítva az 1.1.0 verzióban: a válasz-fertőtlenítő mostantól eltávolítja azokat a nem szabványos mezőket ("x_groq", "usage_breakdown" stb.), amelyek az OpenAI SDK Pydantic ellenőrzési hibáit okozzák -**A GLM/ERNIE elutasítja a "rendszerszerepet"**- Javítva az 1.1.0 verzióban: a szerepnormalizáló automatikusan egyesíti a rendszerüzeneteket felhasználói üzenetekké az inkompatibilis modelleknél -**„fejlesztői” szerepkör nem ismerhető fel**– Javítva az 1.1.0-s verzióban: automatikusan „rendszerré” konvertálva a nem OpenAI-szolgáltatók számára +### Common format issues -- A**`json_schema` nem működik a Geminivel**— Javítva az 1.1.0-s verzióban: a `response_format` most a Gemini `responseMimeType` + `responseSchema` formátumává alakul--- +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- Az automatikus díjkorlát csak az API-kulcs-szolgáltatókra vonatkozik (nem az OAuth-ra/előfizetésre) -- Ellenőrizze, hogy a**Beállítások → Ellenállás → Szolgáltatói profilok**engedélyezve van-e az automatikus díjkorlátozás -- Ellenőrizze, hogy a szolgáltató 429-es állapotkódokat vagy „Retry-After” fejlécet ad-e vissza### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -A szolgáltatói profilok az alábbi beállításokat támogatják: +### Tuning exponential backoff --**Alapkésleltetés**- Kezdeti várakozási idő az első hiba után (alapértelmezett: 1 mp) -**Maximális késleltetés**- Maximális várakozási idő (alapértelmezett: 30 mp) -**Szorzó**- Mennyivel növelhető a késleltetés egy egymást követő hiba esetén (alapértelmezett: 2x)### Anti-thundering herd +Provider profiles support these settings: -Amikor sok egyidejű kérés ér egy korlátozott sebességű szolgáltatót, az OmniRoute mutex + automatikus sebességkorlátozást használ a kérések sorba rendezésére és a lépcsőzetes hibák megelőzésére. Ez automatikus az API-kulcs-szolgáltatók számára.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Egyes OmniRoute-felhasználók az átjárót a RAG- vagy ügynökveremek elé helyezik. Ezekben a beállításokban gyakran látni egy furcsa mintát: az OmniRoute egészségesnek tűnik (a szolgáltatók, az útválasztási profilok rendben vannak, nincs sebességkorlátozási figyelmeztetés), de a végső válasz továbbra is rossz. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -A gyakorlatban ezek az incidensek általában az alsó RAG-csővezetékből származnak, nem pedig magából az átjáróból. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Ha megosztott szókészletet szeretne ezeknek a hibáknak a leírására, használhatja a WFGY ProblemMap-et, egy külső MIT licencszöveg erőforrást, amely tizenhat ismétlődő RAG/LLM hibamintát határoz meg. Magas szinten a következőkre terjed ki: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- visszakeresési sodródás és megtört kontextushatárok -- üres vagy elavult indexek és vektortárak -- beágyazás versus szemantikai eltérés -- azonnali összeszerelési és kontextusablak problémák -- logikai összeomlás és túlságosan magabiztos válaszok -- hosszú láncú és ügynökkoordinációs hibák -- többügynök memória és szerepsodródás -- telepítési és bootstrap rendelési problémák +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -Az ötlet egyszerű: +The idea is simple: -1. Amikor egy rossz választ vizsgál, rögzítse: - - felhasználói feladat és kérés - - útvonal vagy szolgáltató kombinációja az OmniRoute-ban - - bármely lefelé használt RAG kontextus (lekért dokumentumok, eszközhívások stb.) -2. Térképezze le az incidenst egy vagy két WFGY ProblemMap számra (`No.1` … `No.16`). -3. Tárolja a számot saját irányítópultjában, runbookjában vagy eseménykövetőjében az OmniRoute naplók mellett. -4. Használja a megfelelő WFGY oldalt annak eldöntésére, hogy módosítania kell-e a RAG-vermet, a retrievert vagy az útválasztási stratégiát. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -A teljes szöveg és a konkrét receptek itt élnek (MIT licenc, csak szöveg): +Full text and concrete recipes live here (MIT license, text only): [WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -Figyelmen kívül hagyhatja ezt a szakaszt, ha nem futtat RAG- vagy ügynökfolyamatokat az OmniRoute mögött.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**GitHub-problémák**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Architektúra**: A belső részletekért lásd: [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) -**API-referencia**: Lásd: [`docs/API_REFERENCE.md`](API_REFERENCE.md) az összes végponthoz -**Egészségügyi irányítópult**: Az**Irányítópult → Egészség**menüpontban ellenőrizze a valós idejű rendszerállapotot -**Fordító**: Használja az**Irányítópult → Fordító**lehetőséget a formátumhibák elhárításához +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/hu/llm.txt b/docs/i18n/hu/llm.txt new file mode 100644 index 0000000000..39c9afd320 --- /dev/null +++ b/docs/i18n/hu/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Magyar) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Áttekintés + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Biztonság +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/id/README.md b/docs/i18n/id/README.md index 3d46501471..aed0df12fb 100644 --- a/docs/i18n/id/README.md +++ b/docs/i18n/id/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Proksi API universal Anda — satu titik akhir, 60+ penyedia, tanpa waktu henti. Kini dengan**Server MCP (25 alat)**,**Protokol A2A**,**Sistem Memori/Keterampilan**&**Aplikasi Desktop Elektron**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Penyelesaian Obrolan • Penyematan • Pembuatan Gambar • Video • Musik • Audio • Pemeringkatan Ulang •**Penelusuran Web**• Server MCP • Protokol A2A • 100% TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Proksi API universal Anda — satu titik akhir, 60+ penyedia, tanpa waktu henti [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Situs Web](https://omniroute.online) • [🚀 Mulai Cepat](#-mulai cepat) • [💡 Fitur](#-fitur-kunci) • [📖 Dokumen](#-dokumentasi) • [💰 Harga](#-harga-sekilas) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Tersedia dalam:**🇮🇩 [Bahasa Inggris](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇪 [Bahasa Spanyol](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italia](docs/i18n/it/README.md) | 🇷 [Русский](docs/i18n/ru/README.md) | CNY [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Jerman](docs/i18n/de/README.md) | 🇮🇩 [हिन्दी](docs/i18n/in/README.md) | 🇹hon [ไทย](docs/i18n/th/README.md) | 🇮🇩 [Українська](docs/i18n/uk-UA/README.md) | Somalia [العربية](docs/i18n/ar/README.md) | 🇯PH[日本語](docs/i18n/ja/README.md) | 🇻€ [Tiếng Việt](docs/i18n/vi/README.md) | 🇧GET [Български](docs/i18n/bg/README.md) | 🇩 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | ❤ [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | Korea Selatan [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | €🇱 [Nederland](docs/i18n/nl/README.md) | €🇴 [Norsk](docs/i18n/no/README.md) | ppa🇹 [Portugis (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇱 [Polski](docs/i18n/pl/README.md) | 🇲🇾 [Slovenčina](docs/i18n/sk/README.md) | 🇲🇾 [Svenska](docs/i18n/sv/README.md) | كَوَوَوَوَنَ َبَقَوَوَنَوَنَوَوَوَنَوَوَنَوَوَوَونَونَوَوَونَوَـــــــــــــــــِـــــــــــــــــــــــــــــــَــــــــ](docs/i18n/phi/README.md) | Cheska [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,553 +60,629 @@ _Proksi API universal Anda — satu titik akhir, 60+ penyedia, tanpa waktu henti ## 📸 Dashboard Preview - -Klik untuk melihat tangkapan layar dasbor +
+Click to see dashboard screenshots -| Halaman | Tangkapan layar | -| ------------------ | ------------------------------------------------------ | ---------- | -| **Penyedia** | ![Penyedia](docs/screenshot/01-providers.png) | -| **Kombo** | ![Kombo](dokumen/tangkapan layar/02-combos.png) | -| **Analitik** | ![Analytics](dokumen/tangkapan layar/03-analytics.png) | -| **Kesehatan** | ![Kesehatan](docs/screenshot/04-health.png) | -| **Penerjemah** | ![Penerjemah](docs/screenshot/05-translator.png) | -| **Pengaturan** | ![Pengaturan](dokumen/tangkapan layar/06-settings.png) | -| **Alat CLI** | ![Alat CLI](dokumen/tangkapan layar/07-cli-tools.png) | -| **Log Penggunaan** | ![Penggunaan](docs/screenshot/08-usage.png) | -| **Titik Akhir** | ![Titik Akhir](docs/screenshot/09-endpoint.png) |
| +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | + + --- ### 🤖 Free AI Provider for your favorite coding agents -_Hubungkan alat IDE atau CLI apa pun yang didukung AI melalui OmniRoute — gerbang API gratis untuk pengkodean tanpa batas._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ - + - - - - - - - - - - - +
+ OpenClaw
- Buka Cakar + OpenClaw

⭐ 205K
+ NanoBot
NanoBot

- ⭐ 20,9K + ⭐ 20.9K
+ PicoClaw
PicoClaw

- ⭐ 14,6K + ⭐ 14.6K
+ ZeroClaw
ZeroClaw

- ⭐ 9,9K + ⭐ 9.9K
+ IronClaw
- Cakar Besi + IronClaw

- ⭐ 2,1K + ⭐ 2.1K
+ OpenCode
- Kode Terbuka + OpenCode

⭐ 106K
+ Codex CLI
- Kodeks CLI + Codex CLI

- ⭐ 60,8K + ⭐ 60.8K
+ - Kode Claude
- Kode Claude + Claude Code
+ Claude Code

- ⭐ 67,3K + ⭐ 67.3K
+ Gemini CLI
- KLI Gemini + Gemini CLI

- ⭐ 94,7K + ⭐ 94.7K
+ - Kode Kilo
- Kode Kilo + Kilo Code
+ Kilo Code

- ⭐ 15,5K + ⭐ 15.5K
-📡 Semua agen terhubung melalui http://localhost:20128/v1 atau http://cloud.omniroute.online/v1 — satu konfigurasi, model dan kuota tak terbatas--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Berhenti membuang-buang uang dan mencapai batas:** +**Stop wasting money and hitting limits:** -- Kuota berlangganan habis dan tidak terpakai setiap bulan -- Batasan kecepatan menghentikan Anda di tengah-tengah coding -- API mahal ($20-50/bulan per penyedia) -- Peralihan manual antar penyedia +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute memecahkan masalah ini:** +**OmniRoute solves this:** -- ✅**Maksimalkan langganan**- Lacak kuota, gunakan setiap bit sebelum menyetel ulang -- ✅**Pengembalian otomatis**- Berlangganan → Kunci API → Murah → Gratis, tanpa downtime -- ✅**Multi-akun**- Round-robin antar akun per penyedia -- ✅**Universal**- Bekerja dengan Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, alat CLI apa pun--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Bergabunglah dengan komunitas kami!**[Grup WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Dapatkan bantuan, berbagi kiat, dan terus dapatkan informasi terbaru. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Situs Web**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Masalah**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Grup Komunitas](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Berkontribusi**: Lihat [CONTRIBUTING.md](CONTRIBUTING.md), buka PR, atau pilih `terbitan pertama yang bagus` -**Proyek Asli**: [9router oleh decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Saat membuka masalah, jalankan perintah info sistem dan lampirkan file yang dihasilkan:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Ini menghasilkan `system-info.txt` dengan versi Node.js, versi OmniRoute, detail OS, alat CLI yang diinstal (qoder, gemini, claude, codex, antigravity, droid, dll.), status Docker/PM2, dan paket sistem — semua yang kami perlukan untuk mereproduksi masalah Anda dengan cepat. Lampirkan file langsung ke masalah GitHub Anda.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Setiap pengembang yang menggunakan alat AI menghadapi masalah ini setiap hari.**OmniRoute dibuat untuk menyelesaikan semuanya — mulai dari pembengkakan biaya hingga pemblokiran regional, mulai dari aliran OAuth yang rusak hingga operasi protokol dan kemampuan observasi perusahaan. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. - -💸 1. "Saya membayar langganan yang mahal namun tetap terganggu oleh batasan" +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Pengembang membayar $20–200/bulan untuk Claude Pro, Codex Pro, atau GitHub Copilot. Bahkan saat membayar, kuota memiliki batas tertinggi — penggunaan 5 jam, batas mingguan, atau batas tarif per menit. Di tengah sesi pengkodean, penyedia berhenti merespons dan pengembang kehilangan aliran dan produktivitas. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Bagaimana OmniRoute menyelesaikannya:** +**How OmniRoute solves it:** --**Smart 4-Tier Fallback**— Jika kuota berlangganan habis, otomatis dialihkan ke Kunci API → Murah → Gratis tanpa intervensi manual --**Pelacakan Batas Penyedia**— Refresh snapshot kuota yang di-cache pada jadwal sisi server (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) dengan refresh manual tersedia di UI --**Dukungan Multi-Akun**— Beberapa akun per penyedia dengan sistem round-robin otomatis — jika satu akun habis, beralih ke akun berikutnya --**Kombo Kustom**— Rantai fallback yang dapat disesuaikan dengan 9 strategi penyeimbangan (prioritas, tertimbang, pengisian pertama, round-robin, P2C, acak, paling jarang digunakan, optimalisasi biaya, acak ketat) --**Codex Business Quotas**— Pemantauan kuota ruang kerja Bisnis/Tim langsung di dasbor
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard - -🔌 2. "Saya perlu menggunakan beberapa penyedia tetapi masing-masing penyedia memiliki API yang berbeda" + -OpenAI menggunakan satu format, Claude (Anthropic) menggunakan format lain, Gemini menggunakan format lain. Jika pengembang ingin menguji model dari penyedia yang berbeda atau melakukan fallback di antara penyedia tersebut, mereka perlu mengonfigurasi ulang SDK, mengubah titik akhir, menangani format yang tidak kompatibel. Penyedia khusus (FriendLI, NIM) memiliki titik akhir model non-standar. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Bagaimana OmniRoute menyelesaikannya:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Titik Akhir Terpadu**— Satu `http://localhost:20128/v1` berfungsi sebagai proxy untuk 60+ penyedia --**Terjemahan Format**— Otomatis dan transparan: OpenAI ↔ Claude ↔ Gemini ↔ Responses API --**Sanitasi Respons**— Menghapus kolom non-standar (`x_groq`, `usage_breakdown`, `service_tier`) yang merusak OpenAI SDK v1.83+ --**Normalisasi Peran**— Mengonversi `pengembang` → `sistem` untuk penyedia non-OpenAI; `sistem` → `pengguna` untuk GLM/ERNIE --**Think Tag Extraction**— Mengekstrak blok `` dari model seperti DeepSeek R1 ke dalam `reasoning_content` standar --**Output Terstruktur untuk Gemini**— `json_schema` → `responseMimeType`/`responseSchema` konversi otomatis --**`stream` defaultnya adalah `false`**— Sesuai dengan spesifikasi OpenAI, menghindari SSE yang tidak terduga di SDK Python/Rust/Go
+**How OmniRoute solves it:** - -🌐 3. "Penyedia AI saya memblokir wilayah/negara saya" +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Penyedia seperti OpenAI/Codex memblokir akses dari wilayah geografis tertentu. Pengguna mendapatkan error seperti `unsupported_country_region_territory` selama koneksi OAuth dan API. Hal ini sangat membuat frustasi bagi pengembang dari negara-negara berkembang. + -**Bagaimana OmniRoute menyelesaikannya:** +
+🌐 3. "My AI provider blocks my region/country" --**Konfigurasi Proksi 3 Tingkat**— Proksi yang dapat dikonfigurasi pada 3 tingkat: global (semua lalu lintas), per penyedia (hanya satu penyedia), dan per koneksi/kunci --**Lencana Proksi Berkode Warna**— Indikator visual: 🟢 proksi global, 🟡 proksi penyedia, 🔵 proksi koneksi, selalu menampilkan IP --**OAuth Token Exchange Through Proxy**— Alur OAuth juga melewati proxy, menyelesaikan `unsupported_country_region_territory` --**Tes Koneksi melalui Proxy**— Tes koneksi menggunakan proxy yang dikonfigurasi (tidak ada lagi bypass langsung) --**Dukungan SOCKS5**— Dukungan proksi SOCKS5 penuh untuk perutean keluar --**TLS Fingerprint Spoofing**— Sidik jari TLS mirip browser melalui `wreq-js` untuk melewati deteksi bot --**🔏 Pencocokan Sidik Jari CLI**— Menyusun ulang header dan kolom isi agar cocok dengan tanda tangan biner CLI asli, sehingga secara drastis mengurangi risiko penandaan akun. IP proxy dipertahankan — Anda mendapatkan**dan**penyembunyian IP secara bersamaan
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. - -🆓 4. "Saya ingin menggunakan AI untuk coding tapi saya tidak punya uang" +**How OmniRoute solves it:** -Tidak semua orang mampu membayar $20–200/bulan untuk berlangganan AI. Pelajar, pengembang dari negara-negara berkembang, penghobi, dan pekerja lepas memerlukan akses ke model berkualitas tanpa biaya. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Bagaimana OmniRoute menyelesaikannya:** + --**Penyedia Tingkat Gratis Bawaan**— Dukungan asli untuk 100% penyedia gratis: Qoder (5 model tak terbatas melalui OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 model tak terbatas: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID gratis), Gemini CLI (gratis 180 ribu token/bulan) --**Ollama Cloud**— Model Ollama yang dihosting cloud di `api.ollama.com` dengan tingkat "Penggunaan ringan" gratis; gunakan awalan `ollamacloud/` --**Kombo Khusus Gratis**— Rantai `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/bulan tanpa waktu henti --**Akses Gratis NVIDIA NIM**— akses gratis dev-selamanya ~40 RPM ke 70+ model di build.nvidia.com (beralih dari kredit ke batas tarif murni) --**Strategi Pengoptimalan Biaya**— Strategi perutean yang secara otomatis memilih penyedia termurah yang tersedia +
+🆓 4. "I want to use AI for coding but I have no money" - -🔒 5. "Saya perlu melindungi gateway AI saya dari akses tidak sah" +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -Saat mengekspos gateway AI ke jaringan (LAN, VPS, Docker), siapa pun yang memiliki alamat tersebut dapat menggunakan token/kuota pengembang. Tanpa perlindungan, API rentan terhadap penyalahgunaan, injeksi cepat, dan penyalahgunaan. +**How OmniRoute solves it:** -**Bagaimana OmniRoute menyelesaikannya:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**Manajemen Kunci API**— Pembuatan, rotasi, dan pelingkupan per penyedia dengan halaman `/dashboard/api-manager` khusus --**Izin Tingkat Model**— Membatasi kunci API untuk model tertentu (`openai/*`, pola karakter pengganti), dengan tombol Izinkan Semua/Batasi --**API Endpoint Protection**— Memerlukan kunci untuk `/v1/models` dan memblokir penyedia tertentu dari daftar --**Auth Guard + Perlindungan CSRF**— Semua rute dasbor dilindungi dengan middleware `withAuth` + token CSRF --**Pembatas Kecepatan**— Pembatasan kecepatan per-IP dengan jendela yang dapat dikonfigurasi --**Pemfilteran IP**— Daftar yang diizinkan/daftar blokir untuk kontrol akses --**Prompt Injection Guard**— Sanitasi terhadap pola prompt berbahaya --**Enkripsi AES-256-GCM**— Kredensial dienkripsi saat disimpan
+ - -🛑 6. "Penyedia saya down dan saya kehilangan alur coding" +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -Penyedia AI bisa menjadi tidak stabil, menampilkan kesalahan 5xx, atau mencapai batas kecepatan sementara. Jika pengembang bergantung pada satu penyedia, mereka akan terganggu. Tanpa pemutus sirkuit, percobaan ulang yang berulang-ulang dapat membuat aplikasi crash. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Bagaimana OmniRoute menyelesaikannya:** +**How OmniRoute solves it:** --**Pemutus Sirkuit per model**— Buka/tutup otomatis dengan ambang batas dan cooldown yang dapat dikonfigurasi (Tertutup/Terbuka/Setengah Terbuka), tercakup per model untuk menghindari blok berjenjang --**Kemunduran Eksponensial**— Penundaan percobaan ulang yang progresif --**Kawanan Anti-Guntur**— Perlindungan mutex + semaphore terhadap badai percobaan ulang secara bersamaan --**Combo Fallback Chains**— Jika penyedia utama gagal, otomatis gagal dalam rantai tanpa intervensi --**Combo Circuit Breaker**— Menonaktifkan secara otomatis penyedia yang gagal dalam rantai kombo --**Dasbor Kesehatan**— Pemantauan waktu aktif, status pemutus sirkuit, penguncian, statistik cache, latensi p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest - -🔧 7. "Mengonfigurasi setiap alat AI itu membosankan dan berulang-ulang" + -Pengembang menggunakan Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Setiap alat memerlukan konfigurasi yang berbeda (titik akhir API, kunci, model). Mengonfigurasi ulang saat berpindah penyedia atau model hanya membuang-buang waktu. +
+🛑 6. "My provider went down and I lost my coding flow" -**Bagaimana OmniRoute menyelesaikannya:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**Dasbor Alat CLI**— Halaman khusus dengan pengaturan sekali klik untuk Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline --**GitHub Copilot Config Generator**— Menghasilkan `chatLanguageModels.json` untuk VS Code dengan pemilihan model massal --**Onboarding Wizard**— Guided 4-step setup for first-time users --**Satu titik akhir, semua model**— Konfigurasikan `http://localhost:20128/v1` sekali, akses 60+ penyedia
+**How OmniRoute solves it:** - -🔑 8. "Mengelola token OAuth dari beberapa penyedia adalah hal yang buruk" +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot — semuanya menggunakan OAuth 2.0 dengan token yang kedaluwarsa. Pengembang perlu melakukan autentikasi ulang terus-menerus, menangani `rahasia_klien yang hilang`, `redirect_uri_mismatch`, dan kegagalan pada server jarak jauh. OAuth pada LAN/VPS sangat bermasalah. + -**Bagaimana OmniRoute menyelesaikannya:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Penyegaran Token Otomatis**— Penyegaran token OAuth di latar belakang sebelum masa berlakunya habis --**OAuth 2.0 (PKCE) Bawaan**— Aliran otomatis untuk Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder --**OAuth Multi-Akun**— Beberapa akun per penyedia melalui ekstraksi token JWT/ID --**OAuth LAN/Remote Fix**— Deteksi IP pribadi untuk `redirect_uri` + mode URL manual untuk server jarak jauh --**OAuth Dibalik Nginx**— Menggunakan `window.location.origin` untuk kompatibilitas proxy terbalik --**Panduan OAuth Jarak Jauh**— Panduan langkah demi langkah untuk kredensial Google Cloud di VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. - -📊 9. "Saya tidak tahu berapa banyak yang saya belanjakan atau di mana" +**How OmniRoute solves it:** -Pengembang menggunakan beberapa penyedia berbayar tetapi tidak memiliki pandangan terpadu mengenai pembelanjaan. Setiap penyedia memiliki dasbor penagihannya sendiri, namun tidak ada tampilan gabungan. Biaya tak terduga bisa menumpuk. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Bagaimana OmniRoute menyelesaikannya:** + --**Dasbor Analisis Biaya**— Pelacakan biaya per token dan pengelolaan anggaran per penyedia --**Batas Anggaran per Tingkat**— Batas pembelanjaan per tingkat yang memicu penggantian otomatis --**Konfigurasi Harga Per Model**— Harga per model yang dapat dikonfigurasi --**Statistik Penggunaan Per Kunci API**— Jumlah permintaan dan stempel waktu terakhir digunakan per kunci --**Dasbor Analytics**— Kartu statistik, diagram penggunaan model, tabel penyedia dengan tingkat keberhasilan dan latensi +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" - -🐛 10. "Saya tidak bisa mendiagnosis kesalahan dan masalah dalam panggilan AI" +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Saat panggilan gagal, pengembang tidak mengetahui apakah itu batas kecepatan, token kedaluwarsa, format salah, atau kesalahan penyedia. Log terfragmentasi di terminal yang berbeda. Tanpa observabilitas, debugging adalah trial-and-error. +**How OmniRoute solves it:** -**Bagaimana OmniRoute menyelesaikannya:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Dasbor Log Terpadu**— 4 tab: Log Permintaan, Log Proksi, Log Audit, Konsol --**Penampil Log Konsol**— Penampil gaya terminal real-time dengan level kode warna, gulir otomatis, pencarian, filter --**Log Proxy SQLite**— Log persisten yang bertahan saat server dimulai ulang --**Translator Playground**— 4 mode debugging: Playground (terjemahan format), Chat Tester (pulang pergi), Test Bench (batch), Live Monitor (real-time) --**Telemetri Permintaan**— latensi p50/p95/p99 + penelusuran X-Request-Id --**Logging Berbasis File dengan Rotasi**— Log aplikasi dirotasi berdasarkan ukuran, hari penyimpanan, dan jumlah arsip; artefak log panggilan dirotasi berdasarkan hari retensi dan jumlah file --**Laporan Info Sistem**— `npm run system-info` menghasilkan `system-info.txt` dengan lingkungan lengkap Anda (Versi Node, versi OmniRoute, OS, alat CLI, status Docker/PM2). Lampirkan saat melaporkan masalah untuk triase instan.
+ - -🏗️ 11. "Menerapkan dan memelihara gateway itu rumit" +
+📊 9. "I don't know how much I'm spending or where" -Menginstal, mengonfigurasi, dan memelihara proksi AI di berbagai lingkungan (lokal, VPS, Docker, cloud) membutuhkan banyak tenaga. Masalah seperti jalur hardcode, `EACCES` pada direktori, konflik port, dan pembangunan lintas platform menambah gesekan. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Bagaimana OmniRoute menyelesaikannya:** +**How OmniRoute solves it:** --**npm global install**— `npm install -g omniroute && omniroute` — selesai --**Docker Multi-Platform**— asli AMD64 + ARM64 (Apple Silicon, AWS Graviton, Raspberry Pi) --**Docker Compose Profiles**— `base` (tanpa alat CLI) dan `cli` (dengan Claude Code, Codex, OpenClaw) --**Aplikasi Desktop Electron**— Aplikasi asli untuk Windows/macOS/Linux dengan baki sistem, mulai otomatis, mode offline --**Mode Port Terpisah**— API dan Dasbor pada port terpisah untuk skenario tingkat lanjut (proksi terbalik, jaringan kontainer) --**Cloud Sync**— Konfigurasi sinkronisasi antar perangkat melalui Cloudflare Workers --**DB Backups**— Pencadangan otomatis, pemulihan, ekspor dan impor semua pengaturan, dengan `DISABLE_SQLITE_AUTO_BACKUP` untuk pencadangan yang dikelola secara eksternal
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency - -🌍 12. "Antarmukanya hanya berbahasa Inggris dan tim saya tidak bisa berbahasa Inggris" + -Tim di negara-negara yang tidak berbahasa Inggris, khususnya di Amerika Latin, Asia, dan Eropa, kesulitan dengan antarmuka yang hanya berbahasa Inggris. Hambatan bahasa mengurangi adopsi dan meningkatkan kesalahan konfigurasi. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Bagaimana OmniRoute menyelesaikannya:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Dasbor i18n — 30 Bahasa**— 500+ tombol diterjemahkan termasuk Arab, Bulgaria, Denmark, Jerman, Spanyol, Finlandia, Prancis, Ibrani, Hindi, Hungaria, Indonesia, Italia, Jepang, Korea, Melayu, Belanda, Norwegia, Polandia, Portugis (PT/BR), Rumania, Rusia, Slovakia, Swedia, Thailand, Ukraina, Vietnam, China, Filipina, Inggris --**Dukungan RTL**— Dukungan kanan ke kiri untuk bahasa Arab dan Ibrani --**README Multi-Bahasa**— 30 terjemahan dokumentasi lengkap --**Pemilih Bahasa**— Ikon bola dunia di header untuk peralihan waktu nyata
+**How OmniRoute solves it:** - -🔄 13. "Saya memerlukan lebih dari sekedar chat — saya memerlukan embeddings, gambar, audio" +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -AI bukan hanya penyelesaian obrolan. Pengembang perlu membuat gambar, mentranskripsikan audio, membuat penyematan untuk RAG, mengubah peringkat dokumen, dan memoderasi konten. Setiap API memiliki titik akhir dan format yang berbeda. + -**Bagaimana OmniRoute menyelesaikannya:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Embeddings**— `/v1/embeddings` dengan 6 penyedia dan 9+ model --**Image Generation**— `/v1/images/generasi` dengan 10 penyedia dan 20+ model (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Teks-ke-Video**— `/v1/videos/generasi` — ComfyUI (AnimateDiff, SVD) dan SD WebUI --**Teks-ke-Musik**— `/v1/music/generasi` — ComfyUI (Audio Terbuka Stabil, MusicGen) --**Transkripsi Audio**— `/v1/audio/transkripsi` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Text-to-Speech**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + penyedia yang ada --**Moderasi**— `/v1/moderations` — Pemeriksaan keamanan konten --**Pemeringkatan ulang**— `/v1/rerank` — Pemeringkatan ulang relevansi dokumen --**Responses API**— Dukungan penuh `/v1/responses` untuk Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. - -🧪 14. "Saya tidak punya cara untuk menguji dan membandingkan kualitas antar model" +**How OmniRoute solves it:** -Pengembang ingin mengetahui model mana yang terbaik untuk kasus penggunaan mereka — kode, terjemahan, penalaran — tetapi membandingkan secara manual itu lambat. Tidak ada alat evaluasi terintegrasi. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**Bagaimana OmniRoute menyelesaikannya:** + --**Evaluasi LLM**— Pengujian set emas dengan 10 kasus yang dimuat sebelumnya yang mencakup salam, matematika, geografi, pembuatan kode, kepatuhan JSON, terjemahan, penurunan harga, penolakan keamanan --**4 Strategi Pencocokan**— `exact`, `contains`, `regex`, `custom` (fungsi JS) --**Bangku Tes Taman Bermain Penerjemah**— Pengujian batch dengan banyak masukan dan keluaran yang diharapkan, perbandingan lintas penyedia --**Penguji Obrolan**— Perjalanan bolak-balik penuh dengan rendering respons visual --**Monitor Langsung**— Aliran real-time dari semua permintaan yang mengalir melalui proxy +
+🌍 12. "The interface is English-only and my team doesn't speak English" - -📈 15. "Saya perlu melakukan peningkatan tanpa kehilangan performa" +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -Seiring bertambahnya volume permintaan, tanpa menyimpan pertanyaan yang sama akan menghasilkan biaya duplikat. Tanpa idempotensi, permintaan duplikat akan membuang-buang pemrosesan. Batasan tarif per penyedia harus dipatuhi. +**How OmniRoute solves it:** -**Bagaimana OmniRoute menyelesaikannya:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Cache Semantik**— Cache dua tingkat (tanda tangan + semantik) mengurangi biaya dan latensi --**Request Idempoency**— Jendela deduplikasi 5 detik untuk permintaan yang identik --**Deteksi Batas Tarif**— RPM per penyedia, selisih minimum, dan pelacakan serentak maks --**Batas Nilai yang Dapat Diedit**— Default yang dapat dikonfigurasi di Pengaturan → Ketahanan dengan persistensi --**Cache Validasi Kunci API**— cache 3 tingkat untuk kinerja produksi --**Dasbor Kesehatan dengan Telemetri**— latensi p50/p95/p99, statistik cache, waktu aktif
+ - -🤖 16. "Saya ingin mengontrol perilaku model secara global" +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Pengembang yang menginginkan semua respons dalam bahasa tertentu, dengan nada tertentu, atau ingin membatasi token penalaran. Mengonfigurasi ini di setiap alat/permintaan tidak praktis. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Bagaimana OmniRoute menyelesaikannya:** +**How OmniRoute solves it:** --**Injeksi Perintah Sistem**— Perintah global diterapkan ke semua permintaan --**Validasi Anggaran Berpikir**— Kontrol alokasi token penalaran per permintaan (passthrough, otomatis, kustom, adaptif) --**9 Strategi Perutean**— Strategi global yang menentukan cara permintaan didistribusikan --**Wildcard Router**— Pola `penyedia/*` dirutekan secara dinamis ke penyedia mana pun --**Combo Aktifkan/Nonaktifkan Toggle**— Beralih kombo langsung dari dasbor --**Toggle Penyedia**— Mengaktifkan/menonaktifkan semua koneksi untuk penyedia dengan satu klik --**Penyedia yang Diblokir**— Kecualikan penyedia tertentu dari daftar `/v1/models`
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex - -🧰 17. "Saya membutuhkan alat MCP sebagai kemampuan produk kelas satu" + -Banyak gateway AI yang mengekspos MCP hanya sebagai detail implementasi yang tersembunyi. Tim memerlukan lapisan operasi yang terlihat dan dapat dikelola. +
+🧪 14. "I have no way to test and compare quality across models" -**Bagaimana OmniRoute menyelesaikannya:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP muncul di navigasi dasbor dan tab protokol titik akhir -- Halaman manajemen MCP khusus dengan proses, alat, cakupan, dan audit -- Mulai cepat bawaan untuk `omniroute --mcp` dan orientasi klien
+**How OmniRoute solves it:** - -🧠 18. "Saya memerlukan orkestrasi A2A dengan jalur tugas sinkronisasi + streaming" +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Alur kerja agen memerlukan balasan langsung dan eksekusi streaming jangka panjang dengan kontrol siklus hidup. + -**Bagaimana OmniRoute menyelesaikannya:** +
+📈 15. "I need to scale without losing performance" -- Titik akhir A2A JSON-RPC (`POST /a2a`) dengan `message/send` dan `message/stream` -- Streaming SSE dengan propagasi status terminal -- API siklus hidup tugas untuk `tasks/get` dan `tasks/cancel`
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. - -🛰️ 19. "Saya membutuhkan kesehatan proses MCP yang nyata, bukan status yang dapat ditebak" +**How OmniRoute solves it:** -Tim operasional perlu mengetahui apakah MCP benar-benar aktif, bukan hanya apakah API dapat dijangkau. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**Bagaimana OmniRoute menyelesaikannya:** + -- File detak jantung runtime dengan PID, stempel waktu, transportasi, jumlah alat, dan mode cakupan -- API status MCP menggabungkan detak jantung + aktivitas terkini -- Kartu status UI untuk kesegaran proses/waktu aktif/detak jantung +
+🤖 16. "I want to control model behavior globally" - -📋 20. "Saya memerlukan eksekusi alat MCP yang dapat diaudit" +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Saat alat mengubah konfigurasi atau memicu tindakan operasi, tim memerlukan kemampuan penelusuran forensik. +**How OmniRoute solves it:** -**Bagaimana OmniRoute menyelesaikannya:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- Pencatatan audit yang didukung SQLite untuk panggilan alat MCP -- Filter berdasarkan alat, keberhasilan/kegagalan, kunci API, dan penomoran halaman -- Tabel audit dasbor + titik akhir statistik untuk otomatisasi
+ - -🔐 21. "Saya memerlukan izin MCP terbatas untuk setiap integrasi" +
+🧰 17. "I need MCP tools as first-class product capabilities" -Klien yang berbeda harus memiliki akses dengan hak istimewa paling rendah ke kategori alat. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Bagaimana OmniRoute menyelesaikannya:** +**How OmniRoute solves it:** -- 10 cakupan MCP granular untuk akses alat terkontrol -- Penegakan cakupan dan visibilitas di UI manajemen MCP -- Postur default yang aman untuk perkakas operasional
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding - -⚙️ 22. "Saya memerlukan kontrol operasional tanpa melakukan penempatan ulang" + -Tim memerlukan perubahan runtime yang cepat selama insiden atau peristiwa biaya. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**Bagaimana OmniRoute menyelesaikannya:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Beralih aktivasi kombo langsung dari dasbor MCP -- Menerapkan profil ketahanan dari paket kebijakan yang telah ditentukan sebelumnya -- Reset status pemutus sirkuit dari panel operasi yang sama
+**How OmniRoute solves it:** - -🔄 23. "Saya memerlukan visibilitas dan pembatalan siklus hidup tugas A2A langsung" +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Tanpa visibilitas siklus hidup, insiden tugas menjadi sulit untuk diprioritaskan. + -**Bagaimana OmniRoute menyelesaikannya:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Daftar tugas/pemfilteran berdasarkan status/keterampilan dengan penomoran halaman -- Telusuri metadata tugas, peristiwa, dan artefak -- Titik akhir pembatalan tugas dan tindakan UI dengan konfirmasi
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. - -🌊 24. "Saya memerlukan metrik streaming aktif untuk memuat A2A" +**How OmniRoute solves it:** -Alur kerja streaming memerlukan wawasan operasional tentang konkurensi dan koneksi langsung. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**Bagaimana OmniRoute menyelesaikannya:** + -- Penghitung aliran aktif terintegrasi ke dalam status A2A -- Stempel waktu tugas terakhir dan jumlah per negara bagian -- Kartu dasbor A2A untuk pemantauan operasi waktu nyata +
+📋 20. "I need auditable MCP tool execution" - -🪪 25. "Saya memerlukan penemuan agen standar untuk klien" +When tools mutate config or trigger ops actions, teams need forensic traceability. -Klien dan orkestra eksternal memerlukan metadata yang dapat dibaca mesin untuk orientasi. +**How OmniRoute solves it:** -**Bagaimana OmniRoute menyelesaikannya:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- Kartu Agen terekspos di `/.well-known/agent.json` -- Kemampuan dan keterampilan yang ditunjukkan dalam manajemen UI -- API status A2A mencakup metadata penemuan untuk otomatisasi
+ - -🧭 26. "Saya memerlukan kemampuan menemukan protokol di UX produk" +
+🔐 21. "I need scoped MCP permissions per integration" -Jika pengguna tidak dapat menemukan permukaan protokol, kualitas adopsi dan dukungan akan menurun. +Different clients should have least-privilege access to tool categories. -**Bagaimana OmniRoute menyelesaikannya:** +**How OmniRoute solves it:** -- Halaman**Endpoint**terkonsolidasi dengan tab untuk Proxy, MCP, A2A, dan API Endpoints -- Pengalih status layanan inline (Online/Offline) untuk MCP dan A2A -- Tautan dari ikhtisar ke tab manajemen khusus
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling - -🧪 27. "Saya memerlukan validasi protokol end-to-end dengan klien nyata" + -Tes tiruan tidak cukup untuk memvalidasi kompatibilitas protokol sebelum rilis. +
+⚙️ 22. "I need operational controls without redeploying" -**Bagaimana OmniRoute menyelesaikannya:** +Teams need quick runtime changes during incidents or cost events. -- Suite E2E yang mem-boot aplikasi dan menggunakan transportasi klien MCP SDK yang sebenarnya -- Klien A2A menguji penemuan, pengiriman, streaming, dapatkan, dan pembatalan aliran -- Periksa silang pernyataan terhadap audit MCP dan API tugas A2A
+**How OmniRoute solves it:** - -📡 28. "Saya memerlukan kemampuan pengamatan terpadu di semua antarmuka" +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -Memisahkan observabilitas berdasarkan protokol menciptakan titik buta dan MTTR yang lebih panjang. + -**Bagaimana OmniRoute menyelesaikannya:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Dasbor/log/analitik terpadu dalam satu produk -- Kesehatan + audit + permintaan telemetri di seluruh lapisan OpenAI, MCP, dan A2A -- API Operasional untuk status dan otomatisasi
+Without lifecycle visibility, task incidents become hard to triage. - -💼 29. "Saya memerlukan satu runtime untuk proxy + alat + orkestrasi agen" +**How OmniRoute solves it:** -Menjalankan banyak layanan terpisah akan meningkatkan biaya operasional dan mode kegagalan. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**Bagaimana OmniRoute menyelesaikannya:** + -- Proksi yang kompatibel dengan OpenAI, server MCP, dan server A2A dalam satu tumpukan -- Otentikasi bersama, ketahanan, penyimpanan data, dan kemampuan observasi -- Model kebijakan yang konsisten di seluruh platform interaksi +
+🌊 24. "I need active stream metrics for A2A load" - -🚀 30. "Saya perlu mengirimkan alur kerja agen tanpa penyebaran kode lem" +Streaming workflows require operational insight into concurrency and live connections. -Tim kehilangan kecepatan saat menggabungkan beberapa layanan dan skrip ad-hoc. +**How OmniRoute solves it:** -**Bagaimana OmniRoute menyelesaikannya:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Strategi titik akhir terpadu untuk klien dan agen -- UI manajemen protokol bawaan dan jalur validasi asap -- Fondasi siap produksi (keamanan, logging, ketahanan, cadangan)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Playbook A: Maksimalkan langganan berbayar + cadangan murah**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -607,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Playbook B: Tumpukan coding tanpa biaya**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Playbook C: Rantai fallback yang selalu aktif 24/7**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -630,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Playbook D: Operasi agen dengan MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Siapkan pengkodean AI dalam hitungan menit di**$0/bulan**. Hubungkan akun gratis ini dan gunakan kombo**Free Stack**bawaan. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Langkah | Aksi | Penyedia Tidak Terkunci | -| ---- | --------------------------------------------------- | ------------------------------------------------------------------- | -| 1 | Hubungkan**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**tidak terbatas**| -| 2 | Hubungkan**Qoder**(Google OAuth) | pemikiran kimi-k2, qwen3-coder-plus, deepseek-r1... —**tidak terbatas**| -| 3 | Hubungkan**Qwen**(Kode Perangkat) | qwen3-coder-plus, qwen3-coder-flash... —**tidak terbatas**| -| 4 | Hubungkan**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180K/bln gratis**| -| 5 | `/dashboard/combos` →**Templat Tumpukan Gratis ($0)**| Round-robin semua penyedia gratis secara otomatis | +| Step | Action | Providers Unlocked | +| ---- | -------------------------------------------------- | ------------------------------------------------------------------ | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Arahkan IDE/CLI apa pun ke:**`http://localhost:20128/v1` · Kunci API: `any-string` · Selesai. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Cakupan ekstra opsional (juga gratis):**Kunci API Groq (gratis 30 RPM), NVIDIA NIM (gratis 40 RPM, 70+ model), Cerebras (1 juta tok/hari), kunci API LongCat (50 juta token/hari!), Cloudflare Workers AI (10 ribu neuron/hari, 50+ model).## Mulai Cepat +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Mulai Cepat ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **pengguna pnpm:**Jalankan `pnpm Approve-builds -g` setelah instalasi untuk mengaktifkan skrip build asli yang diperlukan oleh `better-sqlite3` dan `@swc/core`: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > -> ```pesta -> instalasi pnpm -g omniroute -> pnpm Approve-builds -g # Pilih semua paket → setujui -> rute omni +> ```bash +> pnpm install -g omniroute +> pnpm approve-builds -g # Select all packages → approve +> omniroute > ``` -Dasbor terbuka di `http://localhost:20128` dan URL dasar API adalah `http://localhost:20128/v1`. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Perintah | Deskripsi | -| --------------------------- | --------------------------------------------------------------- | -| `omnirute` | Mulai server (`PORT=20128`, API dan dasbor pada port yang sama) | -| `omniroute --port 3000` | Setel port kanonik/API ke 3000 | -| `omniroute --mcp` | Mulai server MCP (stdio transport) | -| `omniroute --tidak terbuka` | Jangan buka otomatis browser | -| `omniroute --bantuan` | Tampilkan bantuan | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Mode port terpisah opsional:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -Untuk sebagian besar penerapan, Anda hanya memerlukan: +For most deployments, you only need: -| Variabel | Bawaan | Tujuan | -| ------------------------ | ----------------------------- | ------------------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | -| `STREAM_IDLE_TIMEOUT_MS` | mewarisi `REQUEST_TIMEOUT_MS` | Kesenjangan maksimum antara potongan streaming sebelum OmniRoute membatalkan aliran SSE | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -Kompatibilitas mundur dipertahankan: `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` yang ada, dan var batas waktu per lapisan lainnya masih berfungsi dan menggantikan garis dasar bersama. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Penggantian tingkat lanjut tersedia jika Anda memerlukan kontrol yang lebih baik:| Variabel | Bawaan | Tujuan | -| ---------------------------------------- | ------------------------------------------ | ------------------------------------------------------ | -| `FETCH_TIMEOUT_MS` | mewarisi `REQUEST_TIMEOUT_MS` | Total batas waktu permintaan hulu yang digunakan oleh sinyal pembatalan pengambilan utama | -| `FETCH_HEADERS_TIMEOUT_MS` | mewarisi `FETCH_TIMEOUT_MS` | Batas waktu Undici untuk menerima header respons upstream | -| `FETCH_BODY_TIMEOUT_MS` | mewarisi `FETCH_TIMEOUT_MS` | Batas waktu yang tidak ditentukan antara potongan badan hulu (`0` menonaktifkannya) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Batas waktu koneksi TCP Undici habis | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Batas waktu soket tetap hidup yang tidak aktif | -| `TLS_CLIENT_TIMEOUT_MS` | mewarisi `FETCH_TIMEOUT_MS` | Batas waktu untuk permintaan sidik jari TLS yang dilakukan melalui `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | mewarisi `REQUEST_TIMEOUT_MS` atau `30000` | Batas waktu untuk penerusan proxy `/v1` dari port API ke port dasbor | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `maks(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Batas waktu permintaan masuk di server jembatan API | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Batas waktu header masuk di server jembatan API | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Batas waktu tetap hidup di server jembatan API | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Batas waktu ketidakaktifan soket di server jembatan API (`0` menonaktifkannya) | +Advanced overrides are available if you need finer control: -Jika Anda menjalankan OmniRoute di belakang Nginx, Caddy, Cloudflare, atau proksi terbalik lainnya, pastikan proksi tersebut -waktu tunggu juga lebih tinggi daripada waktu tunggu aliran/pengambilan OmniRoute Anda.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Buka Dasbor → `Penyedia` dan sambungkan setidaknya satu penyedia (OAuth atau kunci API). -2. Buka Dasbor → `Titik Akhir` dan buat kunci API. -3. (Opsional) Buka Dasbor → `Kombo` dan atur rantai cadangan Anda.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Bekerja dengan Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, dan SDK yang kompatibel dengan OpenAI.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (untuk operasi yang digerakkan oleh alat):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -Kemudian sambungkan klien MCP Anda melalui `stdio` dan alat uji seperti: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (untuk alur kerja agen-ke-agen):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -759,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Suite ini memvalidasi alur klien MCP dan A2A yang sebenarnya terhadap aplikasi yang sedang berjalan.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -767,13 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` - -Void Linux (template `xbps-src`) +
+Void Linux (`xbps-src` template) -Untuk pengguna Void Linux, Anda dapat membuat paket asli menggunakan `xbps-src`. Simpan blok ini sebagai `srcpkgs/omniroute/template`:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -785,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -793,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -869,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -880,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute tersedia sebagai image Docker publik di [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Lari cepat:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -890,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**Dengan file lingkungan:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Menggunakan Docker Tulis:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Dukungan dasbor untuk penerapan Docker kini mencakup**Cloudflare Quick Tunnel**sekali klik di `Dashboard → Endpoints`. Aktifkan pertama untuk mengunduh `cloudflared` hanya bila diperlukan, memulai terowongan sementara ke titik akhir `/v1` Anda saat ini, dan menampilkan URL `https://*.trycloudflare.com/v1` yang dihasilkan langsung di bawah URL publik normal Anda. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Catatan: +Notes: -- URL Terowongan Cepat bersifat sementara dan berubah setelah setiap restart. -- Terowongan Cepat tidak dipulihkan secara otomatis setelah OmniRoute atau kontainer dimulai ulang. Aktifkan kembali dari dasbor bila diperlukan. -- Penginstalan terkelola saat ini mendukung Linux, macOS, dan Windows di `x64` / `arm64`. -- Terkelola Quick Tunnels default ke transportasi HTTP/2 untuk menghindari peringatan buffer UDP QUIC yang berisik di lingkungan kontainer yang terbatas. Setel `CLOUDFLARED_PROTOCOL=quic` atau `auto` jika Anda menginginkan transportasi yang berbeda. -- Gambar Docker menggabungkan akar CA sistem dan meneruskannya ke `cloudflared` yang dikelola, yang menghindari kegagalan kepercayaan TLS ketika terowongan melakukan bootstrap di dalam wadah. -- SQLite berjalan dalam mode WAL. `docker stop` harus dibiarkan selesai sehingga OmniRoute dapat memeriksa perubahan terbaru kembali ke `storage.sqlite`. -- File Compose yang dibundel sudah menetapkan masa tenggang penghentian 40 detik. Jika Anda menjalankan image secara langsung, pertahankan `--stop-timeout 40` (atau serupa) sehingga penghentian manual tidak menghentikan pembersihan pematian. -- Setel `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` jika Anda ingin OmniRoute menggunakan biner yang sudah ada alih-alih mengunduhnya. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Menggunakan Docker Compose dengan Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute dapat diekspos dengan aman menggunakan penyediaan SSL otomatis Caddy. Pastikan data DNS A domain Anda mengarah ke IP server Anda.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` - -| Gambar | Tandai | Ukuran | Deskripsi | +| Image | Tag | Size | Description | | ------------------------ | -------- | ------ | --------------------- | -| `diegosouzapw/omniroute` | `terbaru` | ~250MB | Rilis stabil terbaru | -| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Versi saat ini |--- +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | + +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**BARU!**OmniRoute kini tersedia sebagai**aplikasi desktop asli**untuk Windows, macOS, dan Linux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Jalankan OmniRoute sebagai aplikasi desktop mandiri — tanpa terminal, tanpa browser, tanpa internet untuk model lokal. Aplikasi berbasis Electron meliputi: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Jendela Asli**— Jendela aplikasi khusus dengan integrasi baki sistem -- 🔄**Mulai Otomatis**— Luncurkan OmniRoute saat login sistem -- 🔔**Native Notifications**— Get alerts for quota exhaustion or provider issues -- ⚡**Instal Sekali Klik**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Mode Offline**— Bekerja sepenuhnya offline dengan server yang dibundel### Mulai Cepat +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Mulai Cepat ```bash # Development mode @@ -979,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -Saat diminimalkan, OmniRoute ada di baki sistem Anda dengan tindakan cepat: +When minimized, OmniRoute lives in your system tray with quick actions: -- Buka dasbor -- Ubah port server -- Keluar dari aplikasi +- Open dashboard +- Change server port +- Quit application -📖 Dokumentasi lengkap: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Tingkat | Penyedia | Biaya | Reset Kuota | Terbaik Untuk | -| ------------------- | --------------------------- | ---------------------------------- | ------------------------- | ---------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 BERLANGGANAN** | Kode Claude (Pro) | $20/bln | 5 jam + mingguan | Sudah berlangganan | -| | Kodeks (Plus/Pro) | $20-200/bln | 5 jam + mingguan | Pengguna OpenAI | -| | CLI Gemini | **GRATIS** | 180K/bln + 1K/hari | Setiap orang! | -| | Kopilot GitHub | $10-19/bln | Bulanan | Pengguna GitHub | -| **🔑 KUNCI API** | NVIDIA NIM | **GRATIS**(pengembangan selamanya) | ~40 RPM | 70+ model terbuka | -| | Otak | **GRATIS**(1 juta tok/hari) | 60K TPM / 30 RPM | Tercepat di dunia | -| | Bagus | **GRATIS**(30 RPM) | RPD 14,4K | Llama/Gemma ultra-cepat | -| | DeepSeek V3.2 | $0,27/$1,10 per 1 juta | Tidak ada | Alasan harga/kualitas terbaik | -| | xAI Grok-4 Cepat | **$0,20/$0,50 per 1 juta**🆕 | Tidak ada | Panggilan alat + tercepat, sangat rendah | -| | xAI Grok-4 (standar) | $0,20/$1,50 per 1 juta 🆕 | Tidak ada | Penalaran andalan dari xAI | -| | Mistral | Uji coba gratis + berbayar | Tarif terbatas | AI Eropa | -| | BukaRouter | Bayar per penggunaan | Tidak ada | 100+ model agr. | -| **💰 MURAH** | GLM-5 (melalui Z.AI) 🆕 | $0,5/1 juta | Setiap hari pukul 10 pagi | Output 128K, andalan terbaru | -| | GLM-4.7 | $0,6/1 juta | Setiap hari pukul 10 pagi | Cadangan anggaran | -| | MiniMax M2.5 🆕 | masukan $0,3/1 juta | 5 jam bergulir | Penalaran + tugas agen | -| | MiniMax M2.1 | $0,2/1 juta | 5 jam bergulir | Pilihan termurah | -| | Kimi K2.5 (API Moonshot) 🆕 | Bayar per penggunaan | Tidak ada | Akses langsung Moonshot API | -| | Kimi K2 | $9/bln tetap | 10 juta token/bln | Biaya yang dapat diprediksi | -| **🆓 GRATIS** | Qoder | **$0** | Tidak terbatas | 5 model tidak terbatas | -| | Qwen | **$0** | Tidak terbatas | 4 model tidak terbatas | -| | Kiro | **$0** | Tidak terbatas | Claude Soneta/Haiku (Pembuat AWS) | -| | LongCat Flash-Lite 🆕 | **$0**(50 juta tok/hari 🔥) | 1RPS | Kuota gratis terbesar di dunia | -| | Penyerbukan AI 🆕 | **$0**(tidak perlu kunci) | 1 permintaan/15 detik | GPT-5, Claude, DeepSeek, Llama 4 | -| | AI Pekerja Cloudflare 🆕 | **$0**(10K Neuron/hari) | ~150 repetisi/hari | 50+ model, keunggulan global | -| | Scaleway AI 🆕 | **$0**(total 1 juta token) | Tarif terbatas | UE/GDPR, Qwen3 235B, Llama 70B | > 🆕**Model baru ditambahkan (Mar 2026):**Keluarga Grok-4 Fast seharga $0,20/$0,50/M (dibandingkan pada 1143ms — 30% lebih cepat dibandingkan Gemini 2.5 Flash), GLM-5 melalui Z.AI dengan output 128K, penalaran MiniMax M2.5, harga DeepSeek V3.2 yang diperbarui, Kimi K2.5 melalui API langsung Moonshot. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 Tumpukan Kombo $0 — Penyiapan Gratis Lengkap:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Tanpa biaya. Jangan pernah berhenti melakukan pengkodean.**Konfigurasikan ini sebagai satu kombo OmniRoute dan semua fallback terjadi secara otomatis — tidak pernah ada peralihan manual.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Semua model di bawah**100% gratis tanpa memerlukan kartu kredit**. OmniRoute melakukan rute otomatis di antara keduanya ketika satu kuota habis — gabungkan semuanya untuk kombo $0 yang tidak dapat dipecahkan.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Model | Awalan | Batasi | Batas Tarif | +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | --------------------- | -| `claude-soneta-4.5` | `kr/` |**Tidak terbatas**| Tidak ada batas harian yang dilaporkan | -| `claude-haiku-4.5` | `kr/` |**Tidak terbatas**| Tidak ada batas harian yang dilaporkan | -| `claude-opus-4.6` | `kr/` |**Tidak terbatas**| Opus Terbaru melalui Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -| Model | Awalan | Batasi | Batas Tarif | -| ---- | ------ | ------------- | --------------- | -| `kimi-k2-berpikir` | `jika/` |**Tidak terbatas**| Tidak ada batas yang dilaporkan | -| `qwen3-coder-plus` | `jika/` |**Tidak terbatas**| Tidak ada batas yang dilaporkan | -| `pencarian mendalam-r1` | `jika/` |**Tidak terbatas**| Tidak ada batas yang dilaporkan | -| `minimax-m2.1` | `jika/` |**Tidak terbatas**| Tidak ada batas yang dilaporkan | -| `kimi-k2` | `jika/` |**Tidak terbatas**| Tidak ada batas yang dilaporkan | +### 🟢 QODER MODELS (Free PAT via qodercli) -> Metode koneksi yang disarankan:**Token Akses Pribadi + `qodercli`**. Peramban OAuth adalah -> eksperimental dan dinonaktifkan secara default kecuali variabel lingkungan `QODER_OAUTH_*` dikonfigurasi.### 🟡 QWEN MODELS (Device Code Auth) +| Model | Prefix | Limit | Rate Limit | +| ------------------ | ------ | ------------- | --------------- | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -| Model | Awalan | Batasi | Batas Tarif | +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. + +### 🟡 QWEN MODELS (Device Code Auth) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | ------------------- | -| `qwen3-coder-plus` | `qw/` |**Tidak terbatas**| Tidak ada batas yang dilaporkan | -| `qwen3-coder-flash` | `qw/` |**Tidak terbatas**| Tidak ada batas yang dilaporkan | -| `qwen3-coder-berikutnya` | `qw/` |**Tidak terbatas**| Tidak ada batas yang dilaporkan | -| `model visi` | `qw/` |**Tidak terbatas**| Multimodal (gambar) |### 🟣 GEMINI CLI (Google OAuth) +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Model | Awalan | Batasi | Batas Tarif | +### 🟣 GEMINI CLI (Google OAuth) + +| Model | Prefix | Limit | Rate Limit | | ------------------------ | ------ | --------------------------- | ------------- | -| `tinjauan-gemini-3-flash` | `gc/` |**180rb tok/bulan**+ 1rb/hari | Reset bulanan | -| `gemini-2.5-pro` | `gc/` | 180K/bulan (kolam bersama) | Kualitas tinggi |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -| Tingkat | Batas Harian | Batas Tarif | Catatan | +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---------- | ------------ | ----------- | ------------------------------------------------------ | -| Gratis (Pengembangan) | Tidak ada batasan token |**~40 RPM**| 70+ model; transisi ke batas tarif murni pada pertengahan tahun 2025 | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -Model gratis populer: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -| Tingkat | Batas Harian | Batas Tarif | Catatan | +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) + +| Tier | Daily Limit | Rate Limit | Notes | | ---- | ----------------- | ---------------- | ------------------------------------------- | -| Gratis |**1 juta token/hari**| 60K TPM / 30 RPM | Inferensi LLM tercepat di dunia; disetel ulang setiap hari | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -Tersedia gratis: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -| Tingkat | Batas Harian | Batas Tarif | Catatan | +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---- | ------------- | ---------------- | ----------------------------------------- | -| Gratis |**RPD 14,4K**| 30 RPM per model | Tidak ada kartu kredit; 429 pada batas, tidak dikenakan biaya | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -Tersedia gratis: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -| Model | Awalan | Kuota Gratis Harian | Catatan | +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 + +| Model | Prefix | Daily Free Quota | Notes | | ----------------------------- | ------ | ----------------- | ----------------------- | -| `LongCat-Flash-Lite` | `lc/` |**50 juta token**💥 | Kuota gratis terbesar yang pernah ada | -| `Obrolan-Flash-LongCat` | `lc/` | 500 ribu token | Obrolan multi-putaran | -| `Pemikiran-Flash-LongCat` | `lc/` | 500 ribu token | Penalaran / CoT | -| `Pemikiran-Flash-LongCat-2601` | `lc/` | 500 ribu token | Versi Jan 2026 | -| `LongCat-Flash-Omni-2603` | `lc/` | 500 ribu token | Multimoda | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | -> 100% gratis saat dalam versi beta publik. Daftar di [longcat.chat](https://longcat.chat) dengan email atau telepon. Reset setiap hari pukul 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. -| Model | Awalan | Batas Tarif | Penyedia Dibalik | -| ---------- | ------ | ---------- | ---- | -| `buka` | `pol/` | 1 permintaan/15 detik | GPT-5 | -| `claude` | `pol/` | 1 permintaan/15 detik | Claude Antropik | -| `gemini` | `pol/` | 1 permintaan/15 detik | Google Gemini | -| `mencari lebih dalam` | `pol/` | 1 permintaan/15 detik | Pencarian Dalam V3 | -| `lama` | `pol/` | 1 permintaan/15 detik | Meta Llama 4 Pramuka | -| `mistral` | `pol/` | 1 permintaan/15 detik | AI Mistral | +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 -> ✨**Tanpa gesekan:**Tanpa pendaftaran, tanpa kunci API. Tambahkan penyedia Penyerbukan dengan bidang kunci kosong dan itu langsung berfungsi.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +| Model | Prefix | Rate Limit | Provider Behind | +| ---------- | ------ | ---------- | ------------------ | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -| Tingkat | Neuron Harian | Penggunaan Setara | Catatan | +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. + +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 + +| Tier | Daily Neurons | Equivalent Usage | Notes | | ---- | ------------- | --------------------------------------- | ----------------------- | -| Gratis |**10.000**| ~150 respons LLM / audio 500 detik / penyematan 15K | Keunggulan global, 50+ model | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -Model gratis populer: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (audio gratis!), `@cf/qwen/qwen2.5-coder-15b-instruct` +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -> Membutuhkan Token API + ID Akun dari [dash.cloudflare.com](https://dash.cloudflare.com). Simpan ID Akun di pengaturan penyedia.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -| Tingkat | Kuota Gratis | Lokasi | Catatan | +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | | ---- | ------------- | ------------ | ----------------------------------- | -| Gratis |**1 juta token**| 🇫🇷 Paris, UE | Tidak diperlukan kartu kredit dalam batasan | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | -Tersedia gratis: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` -> Sesuai dengan UE/GDPR. Dapatkan kunci API di [console.scaleway.com](https://console.scaleway.com). +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). ->**💡 Tumpukan Gratis Terbaik (11 Penyedia, $0 Selamanya):** +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Soneta/Haiku TANPA BATAS -> Qoder (jika/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 TANPA BATAS -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 juta token/hari 🔥 -> Penyerbukan (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — tidak perlu kunci -> Qwen (qw/) → model qwen3-coder TANPA BATAS -> Gemini (gemini/) → Gemini 2.5 Flash — 1.500 permintaan/hari gratis -> Cloudflare AI (cf/) → 50+ model — 10 ribu Neuron/hari -> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1 juta token gratis (UE) -> Groq (groq/) → Llama/Gemma — 14,4 ribu permintaan/hari sangat cepat -> NVIDIA NIM (nvidia/) → 70+ model terbuka — 40 RPM selamanya -> Otak (otak/) → Llama/Qwen tercepat di dunia — 1 juta tok/hari -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Transkripsikan audio/video apa pun seharga**$0**— Deepgram memimpin dengan $200 gratis, penggantian AssemblyAI $50, Groq Whisper sebagai cadangan darurat tanpa batas. +## 🎙️ Free Transcription Combo -| Penyedia | Kredit Gratis | Model Terbaik | Batas Tarif | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. + +| Provider | Free Credits | Best Model | Rate Limit | | ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | -| 🟢**Deepgram**|**$200 gratis**(mendaftar) | `nova-3` — akurasi terbaik, 30+ bahasa | Tidak ada batasan RPM pada kredit gratis | -| 🔵**PerakitanAI**|**$50 gratis**(mendaftar) | `universal-3-pro` — bab, sentimen, PII | Tidak ada batasan RPM pada kredit gratis | -| 🔴**Groq**|**Gratis selamanya**| `bisikan-besar-v3` — Bisikan OpenAI | 30 RPM (tarif terbatas) | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | -**Kombo yang disarankan di `/dashboard/combos`:**``` +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Kemudian di `/dashboard/media` → tab**Transkripsi**: unggah file audio atau video apa pun → pilih titik akhir kombo Anda → dapatkan transkripsi dalam format yang didukung.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 dibangun sebagai platform operasional, bukan hanya proxy relai.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Fitur | Apa Fungsinya | -| -------------------------------------- | ----------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Keluarga Cepat Grok-4** | model xAI seharga $0,20/$0,50/M — dengan benchmark 1143ms (30% lebih cepat dari Gemini 2.5 Flash) | -| 🧠**GLM-5 melalui Z.AI** | Konteks keluaran 128 ribu, $0,5/1 juta — andalan terbaru dari keluarga GLM | -| 🔮**MiniMax M2.5** | Penalaran + tugas agen seharga $0,30/1 juta — peningkatan signifikan dari M2.1 | -| 🎯**alat Memanggil Bendera per Model** | `toolCalling: true/false` per model di registri — AutoCombo melewatkan model yang tidak mendukung alat | -| 🌍**Deteksi Niat Multibahasa** | Kata kunci PT/ZH/ES/AR dalam penilaian AutoCombo — pemilihan model yang lebih baik untuk konten non-Inggris | -| 📊**Fallback Berdasarkan Tolok Ukur** | Latensi p95 nyata dari penilaian kombo umpan permintaan langsung — AutoCombo belajar dari data aktual | -| 🔁**Minta Deduplikasi** | Jendela dedup berbasis hash konten — aman untuk multi-agen, mencegah biaya duplikat | -| 🔌**Strategi Router Pluggable** | Antarmuka `RouterStrategy` yang dapat diperluas — tambahkan logika perutean khusus sebagai plugin | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Fitur | Apa Fungsinya | -| ---------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Model Taman Bermain** | Halaman dasbor untuk menguji model apa pun secara langsung — pemilih penyedia/model/titik akhir, Editor Monaco, streaming, batalkan, pengaturan waktu | -| 🔏**Pencocokan Sidik Jari CLI** | Pengurutan header/isi per penyedia agar sesuai dengan tanda tangan CLI asli — alihkan per penyedia di Pengaturan > Keamanan.**IP proxy Anda dipertahankan** | -| 🤝**Dukungan ACP (Protokol Klien Agen)** | Penemuan agen CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 lainnya), proses spawner, titik akhir `/api/acp/agents` | -| 🤖**Dasbor Agen ACP** | Debug › Halaman agen — kisi 14 agen dengan status pemasangan, versi, formulir agen khusus untuk alat CLI apa pun. Pengguna**OpenCode**mendapatkan tombol "Unduh opencode.json" yang secara otomatis menghasilkan konfigurasi siap pakai dengan semua model yang tersedia. | -| 🔧**Perutean `apiFormat` Model Kustom** | Model khusus dengan `apiFormat: "responses"` sekarang dirutekan dengan benar ke penerjemah Responses API | -| 🏢**Isolasi Ruang Kerja Codex** | Beberapa ruang kerja Codex per email — OAuth memisahkan koneksi dengan benar berdasarkan ID ruang kerja | -| 🔄**Pembaruan Otomatis Elektron** | Aplikasi desktop memeriksa pembaruan + instal otomatis saat restart | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Fitur | Apa Fungsinya | -| ----------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**Server MCP (25 alat)** | Alat IDE/agen melalui 3 transport: stdio, SSE (`/api/mcp/sse`), HTTP yang dapat dialirkan (`/api/mcp/stream`). 18 inti + 3 memori + 4 alat keterampilan | -| 🤝**Server A2A (JSON-RPC + SSE)** | Eksekusi tugas agen-ke-agen dengan aliran sinkronisasi dan streaming | -| 🧭**Halaman Titik Akhir Konsolidasi** | Halaman manajemen bertab dengan tab Endpoint Proxy, MCP, A2A, dan API Endpoints | -| 🎚️**Tombol Aktifkan/Nonaktifkan Layanan** | Sakelar ON/OFF untuk MCP dan A2A dengan pengaturan persistensi (default: OFF) | -| 🛰️**Detak Jantung Waktu Proses MCP** | Status proses nyata (pid, uptime, usia detak jantung, transportasi, mode cakupan) | -| 📋**Jejak Audit MCP** | Log audit yang dapat difilter dengan keberhasilan/kegagalan dan atribusi kunci | -| 🔐**Penegakan Lingkup MCP** | 10 izin cakupan terperinci untuk akses alat terkontrol | -| 📡**Manajemen Siklus Hidup Tugas A2A** | Daftar/filter tugas, periksa peristiwa/artefak, batalkan tugas yang sedang berjalan | -| 📋**Penemuan Kartu Agen** | `/.well-known/agent.json` untuk penemuan otomatis klien | -| 🧪**Memanfaatkan Uji Protokol E2E** | Alur klien MCP SDK + A2A yang sebenarnya di `test:protocols:e2e` | -| ⚙️**Kontrol Operasional** | Ganti kombo, terapkan profil ketahanan, setel ulang pemutus dari satu permukaan kontrol | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Fitur | Apa Fungsinya | -| ----------------------------------- | ----------------------------------------------------------------------------------------- | ----------------------- | -| 🎯**Pengembalian 4 Tingkat Cerdas** | Rute otomatis: Berlangganan → Kunci API → Murah → Gratis | -| 📊**Pelacakan Kuota Waktu Nyata** | Jumlah token langsung + setel ulang hitungan mundur per penyedia | -| 🔄**Terjemahan Format** | OpenAI ↔ Claude ↔ Gemini ↔ Respons dengan konversi aman skema | -| 👥**Dukungan Multi-Akun** | Banyak akun per penyedia dengan pilihan cerdas | -| 🔄**Penyegaran Token Otomatis** | Token OAuth disegarkan secara otomatis dengan coba lagi | -| 🎨**Kombo Khusus** | 9 strategi penyeimbangan + kontrol rantai mundur | -| 🌐**Router Wildcard** | perutean dinamis `penyedia/*` | -| 🧠**Memikirkan Kontrol Anggaran** | Batas penalaran passthrough, otomatis, kustom, dan adaptif | -| 🔀**Model Alias** | Aliasing model kustom + bawaan dan keamanan migrasi | -| ⚡**Degradasi Latar Belakang** | Arahkan tugas latar belakang berprioritas rendah ke model yang lebih murah | -| 🧪**Perutean Cerdas Sadar Tugas** | Pilih model secara otomatis berdasarkan jenis konten (pengkodean/visi/analisis/ringkasan) | -| 🔄**Alur Kerja Agen A2A** | Orkestra FSM deterministik untuk eksekusi agen multi-langkah stateful | -| 🔀**Perutean Adaptif** | Penggantian strategi dinamis berdasarkan volume token dan kompleksitas yang cepat | -| 🎲**Keberagaman Penyedia** | Penilaian entropi Shannon menyeimbangkan distribusi lalu lintas kombo otomatis | -| 💬**Injeksi Perintah Sistem** | Pengendalian perilaku global diterapkan secara konsisten | -| 📄**Kompatibilitas API Respons** | Dukungan penuh `/v1/responses` untuk Codex dan alur kerja agen tingkat lanjut | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Fitur | Apa Fungsinya | -| ------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**Pembuatan Gambar** | `/v1/images/generasi` dengan cloud dan backend lokal | -| 📐**Sematan** | `/v1/embeddings` untuk saluran pencarian dan RAG | -| 🎤**Transkripsi Audio** | `/v1/audio/transcriptions` — 7 penyedia (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), deteksi bahasa otomatis, dukungan MP4/MP3/WAV | -| 🔊**Teks-ke-Ucapan** | `/v1/audio/speech` — 10 penyedia (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) dengan pesan kesalahan yang benar | -| 🎬**Pembuatan Video** | `/v1/videos/generasi` (Alur kerja ComfyUI + SD WebUI) | -| 🎵**Generasi Musik** | `/v1/music/generasi` (alur kerja ComfyUI) | -| 🛡️**Moderasi** | pemeriksaan keamanan `/v1/moderations` | -| 🔀**Pemeringkatan Ulang** | `/v1/rerank` untuk penilaian relevansi | -| 🔍**Penelusuran Web**🆕 | `/v1/search` — 5 penyedia (Serper, Brave, Perplexity, Exa, Tavily), 6.500+ gratis/bulan, failover otomatis, cache | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Fitur | Apa Fungsinya | -| --------------------------------------------- | --------------------------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**Pemutus Arus** | Perjalanan/pemulihan per model dengan kontrol ambang batas | -| 🎯**Model Sadar Titik Akhir** | Model khusus mendeklarasikan titik akhir yang didukung + format API | -| 🛡️**Kawanan Anti Guntur** | Perlindungan mutex + semaphore pada acara coba ulang/nilai | -| 🧠**Semantik + Cache Tanda Tangan** | Pengurangan biaya/latensi dengan dua lapisan cache | -| ⚡**Minta Idempotensi** | Jendela perlindungan duplikat | -| 🔒**Spoofing Sidik Jari TLS** | Sidik jari TLS seperti browser —**mengurangi deteksi bot dan penandaan akun** | -| 🔏**Pencocokan Sidik Jari CLI** | Cocok dengan tanda tangan permintaan CLI asli —**mengurangi risiko larangan sekaligus mempertahankan IP proxy** | -| 🌐**Pemfilteran IP** | Kontrol daftar yang diizinkan/daftar blokir untuk penerapan yang terbuka | -| 📊**Batas Tarif yang Dapat Diedit** | Batas tingkat global/penyedia yang dapat dikonfigurasi dengan persistensi | -| 📉**Degradasi yang Anggun** | Fallback kemampuan multi-lapis yang melindungi operasi gateway inti | -| 📜**Konfigurasi Jejak Audit** | Pelacakan perubahan berbasis diff mencegah penyimpangan operasional dengan rollback sederhana | -| ⏳**Sinkronisasi Kesehatan Penyedia** | Pemantauan kedaluwarsa token proaktif memicu peringatan sebelum kegagalan otorisasi | -| 🚪**Nonaktifkan Otomatis Akun yang Diblokir** | Pemutus sirkuit operasional menyegel akun token yang diblokir secara permanen secara otomatis | -| 🔑**Manajemen Kunci API + Pelingkupan** | Mengamankan penerbitan/rotasi kunci dan kontrol model/penyedia | -| 👁️**Pengungkapan Kunci API Tercakup**🆕 | Ikut serta dalam pemulihan kunci API melalui `ALLOW_API_KEY_REVEAL` | -| 🛡️**`/model`**yang dilindungi | Gerbang autentikasi opsional dan penyembunyian penyedia untuk katalog model | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Fitur | Apa Fungsinya | -| ------------------------------------ | ---------------------------------------------------------------------- | ---------------------------- | -| 📝**Permintaan + Pencatatan Proksi** | Permintaan/tanggapan lengkap dan pencatatan proksi | -| 📉**Streaming Log Terperinci**🆕 | Merekonstruksi aliran muatan SSE dengan rapi ke dalam UI | -| 📋**Dasbor Log Terpadu** | Tampilan permintaan, proksi, audit, dan konsol dalam satu halaman | -| 🔍**Permintaan Telemetri** | latensi p50/p95/p99 dan pelacakan permintaan | -| 🏥**Dasbor Kesehatan** | Waktu aktif, status pemutus, penguncian, statistik cache | -| 💰**Pelacakan Biaya** | Kontrol anggaran dan visibilitas harga per model | -| 📈**Visualisasi Analisis** | Wawasan penggunaan model/penyedia dan tampilan tren | -| 🧪**Kerangka Evaluasi** | Pengujian set emas dengan strategi pencocokan yang dapat dikonfigurasi | -| 📡**Diagnostik Langsung**🆕 | Bypass cache semantik untuk pengujian langsung kombo yang akurat | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Fitur | Apa Fungsinya | -| --------------------------------- | ------------------------------------------------------------------------------ | --------------------- | -| 🌐**Terapkan Di Mana Saja** | Localhost, VPS, Docker, Lingkungan Cloud | -| 🚇**Terowongan Cloudflare**🆕 | Integrasi Quick Tunnel sekali klik dari dasbor | -| 🔑**Pemfilteran Model Kunci API** | Respons asli /v1/models difilter melalui peran konteks Pembawa yang ditetapkan | -| ⚡**Melewati Cache Cerdas** | Heuristik TTL yang dapat dikonfigurasi dan kontrol pengambilan ulang paksa | -| 🔄**Cadangan/Pemulihan** | Arus ekspor/impor dan pemulihan bencana | -| 🧙**Wizard Orientasi** | Penyiapan terpandu yang dijalankan pertama kali | -| 🔧**Dasbor Alat CLI** | Pengaturan sekali klik untuk alat pengkodean populer | -| 🎮**Model Taman Bermain** | Uji penyedia/model/titik akhir apa pun dari dasbor | -| 🔏**Tolak Sidik Jari CLI** | Pencocokan sidik jari per penyedia di Pengaturan > Keamanan | -| 🌐**i18n (30 bahasa)** | Dasbor lengkap + dukungan bahasa dokumen dengan cakupan RTL | -| 🧹**Hapus Semua Model** | Pembersihan daftar model sekali klik di detail penyedia | -| 👁️**Kontrol Bilah Sisi**🆕 | Sembunyikan komponen dan integrasi dari Pengaturan Penampilan | -| 📋**Template Masalah** | Templat GitHub standar untuk bug dan fitur | -| 📂**Direktori Data Khusus** | Penggantian `DATA_DIR` untuk lokasi penyimpanan | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1292,103 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -Ketika kuota, tarif, atau kesehatan gagal, OmniRoute secara otomatis berpindah ke kandidat berikutnya tanpa peralihan manual.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A dapat ditemukan di UI dan dokumen (tidak disembunyikan) -- API status protokol mengekspos data operasional langsung (`/api/mcp/*`, `/api/a2a/*`) -- Dasbor mencakup tindakan untuk operasi hari ke-2 (pengalihan kombo, pengaturan ulang pemutus, pembatalan tugas)#### Translator + validation workflow +#### Protocol management that is visible and operable -Area Penerjemah meliputi: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Taman Bermain**: meminta pemeriksaan transformasi -**Penguji Obrolan**: permintaan/respons penuh pulang pergi -**Test Bench**: beberapa kasus sekaligus -**Monitor Langsung**: tampilan lalu lintas waktu nyata +#### Translator + validation workflow -Ditambah validasi protokol dengan klien nyata melalui `npm run test:protocols:e2e`. +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Referensi alat, konfigurasi IDE, dan contoh klien +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A Server README](src/lib/a2a/README.md)**— Keterampilan, metode JSON-RPC, streaming, dan siklus hidup tugas## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute menyertakan kerangka evaluasi bawaan untuk menguji kualitas respons LLM terhadap rangkaian emas. Akses melalui**Analytics → Evals**di dasbor.### Built-in Golden Set +## 🧪 Evaluations (Evals) -"OmniRoute Golden Set" yang dimuat sebelumnya berisi kasus uji untuk: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Salam, matematika, geografi, pembuatan kode -- Kepatuhan format JSON, terjemahan, pembuatan penurunan harga -- Penolakan keamanan (konten berbahaya), penghitungan, logika boolean### Evaluation Strategies +### Built-in Golden Set -| Strategi | Deskripsi | Contoh | -| ----------- | ------------------------------------------------------------ | ------------------------------------- | --- | -| `tepat` | Output harus sama persis | `"4"` | -| `berisi` | Output harus berisi substring (tidak peka huruf besar-kecil) | `"Paris"` | -| `regex` | Output harus sesuai dengan pola regex | `"1.*2.*3"` | -| `kebiasaan` | Fungsi JS khusus mengembalikan benar/salah | `(keluaran) => keluaran.panjang > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) - -🧩 Penyiapan MCP (Protokol Konteks Model) +
+🧩 MCP Setup (Model Context Protocol) -Mulai transportasi MCP dalam mode stdio:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Alur validasi yang disarankan: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Hubungkan klien MCP Anda melalui stdio. -2. Jalankan `omniroute_get_health`. -3. Jalankan `omniroute_list_combos`. -4. Buka `/dashboard/mcp` untuk mengonfirmasi detak jantung, aktivitas, dan audit. +Useful APIs for automation: -API yang berguna untuk otomatisasi: +- `GET /api/mcp/status` +- `GET /api/mcp/tools` +- `GET /api/mcp/audit` +- `GET /api/mcp/audit/stats` -- `DAPATKAN /api/mcp/status` -- `DAPATKAN /api/mcp/alat` -- `DAPATKAN /api/mcp/audit` -- `DAPATKAN /api/mcp/audit/statistik`
+ - -🤝 Penyiapan A2A (Agent2Agent) +
+🤝 A2A Setup (Agent2Agent) -Temukan agennya:```bash +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Kirim tugas:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` +Manage lifecycle: -Kelola siklus hidup: +- `GET /api/a2a/status` +- `GET /api/a2a/tasks` +- `GET /api/a2a/tasks/:id` +- `POST /api/a2a/tasks/:id/cancel` -- `DAPATKAN /api/a2a/status` -- `DAPATKAN /api/a2a/tugas` -- `DAPATKAN /api/a2a/tasks/:id` -- `POST /api/a2a/tasks/:id/batal` +Operational UI: -UI Operasional: +- `/dashboard/a2a` for task/state/stream observability and smoke actions -- `/dashboard/a2a` untuk observasi tugas/status/aliran dan tindakan asap
+ - -🧪 Validasi protokol menyeluruh +
+🧪 End-to-end protocol validation -Validasi kedua protokol dengan klien nyata:```bash +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Ini memverifikasi: +This verifies: -- Koneksi/daftar/panggilan klien MCP SDK -- Penemuan A2A/kirim/aliran/dapatkan/batalkan -- Periksa silang data dalam audit MCP dan API manajemen tugas A2A
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs - -💳 Penyedia Langganan### Claude Code (Pro/Max) + + +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1401,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Kiat Pro:**Gunakan Opus untuk tugas kompleks, Soneta untuk kecepatan. OmniRoute melacak kuota per model!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1415,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Setiap akun Codex sekarang memiliki kebijakan yang dapat diubah di `Dasbor -> Penyedia`: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (ON/OFF): menerapkan kebijakan ambang jendela 5 jam. -- `Mingguan` (ON/OFF): menerapkan kebijakan ambang jendela mingguan. -- Perilaku ambang batas: ketika jendela yang diaktifkan mencapai >=90% penggunaan, akun tersebut dilewati. -- Perilaku rotasi: OmniRoute merutekan ke akun Codex berikutnya yang memenuhi syarat secara otomatis. -- Perilaku reset: ketika waktu `resetAt` penyedia telah berlalu, akun akan memenuhi syarat lagi secara otomatis. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Skenario: +Scenarios: -- `5h ON` + `Weekly ON`: akun dilewati ketika salah satu jendela mencapai ambang batas. -- `5h OFF` + `Weekly ON`: hanya penggunaan mingguan yang dapat memblokir akun. -- `5h ON` + `Weekly OFF`: hanya penggunaan 5 jam yang dapat memblokir akun. -- `resetAt` lolos: akun masuk kembali ke rotasi secara otomatis (tidak ada pengaktifan ulang secara manual).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1440,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Nilai Terbaik:**Tingkat gratis yang sangat besar! Gunakan ini sebelum tingkatan berbayar.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1455,71 +1662,91 @@ Models:
- -🔑 Penyedia Kunci API### NVIDIA NIM (FREE developer access — 70+ models) +
+🔑 API Key Providers -1. Daftar: [build.nvidia.com](https://build.nvidia.com) -2. Dapatkan kunci API gratis (termasuk 1000 kredit inferensi) -3. Dasbor → Tambah Penyedia → NVIDIA NIM: - - Kunci API: `nvapi-kunci-Anda` +### NVIDIA NIM (FREE developer access — 70+ models) -**Model:**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, dan 50+ lainnya +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Kiat Pro:**API yang kompatibel dengan OpenAI — bekerja secara lancar dengan terjemahan format OmniRoute!### DeepSeek +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -1. Daftar: [platform.deepseek.com](https://platform.deepseek.com) -2. Dapatkan kunci API -3. Dasbor → Tambah Penyedia → DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -**Model:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +### DeepSeek -1. Daftar: [console.groq.com](https://console.groq.com) -2. Dapatkan kunci API (termasuk tingkat gratis) -3. Dasbor → Tambah Penyedia → Groq +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -**Model:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Kiat Pro:**Inferensi ultra-cepat — terbaik untuk pengkodean waktu nyata!### OpenRouter (100+ Models) +### Groq (Free Tier Available!) -1. Daftar: [openrouter.ai](https://openrouter.ai) -2. Dapatkan kunci API -3. Dasbor → Tambah Penyedia → OpenRouter +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -**Model:**Akses 100+ model dari semua penyedia utama melalui satu kunci API. +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Perilaku dasbor:**Model OpenRouter dikelola dari**Model yang Tersedia**. Penambahan manual, impor, dan sinkronisasi otomatis semuanya memperbarui daftar yang sama.
+**Pro Tip:** Ultra-fast inference — best for real-time coding! - -💰 Penyedia Murah (Cadangan)### GLM-4.7 (Daily reset, $0.6/1M) +### OpenRouter (100+ Models) -1. Daftar: [Zhipu AI](https://open.bigmodel.cn/) -2. Dapatkan kunci API dari Coding Plan -3. Dasbor → Tambahkan Kunci API: - - Penyedia: `glm` - - Kunci API: `kunci Anda` +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -**Gunakan:**`glm/glm-4.7` +**Models:** Access 100+ models from all major providers through a single API key. -**Tips Pro:**Paket Coding menawarkan 3× kuota dengan biaya 1/7! Reset setiap hari pukul 10.00.### MiniMax M2.1 (5h reset, $0.20/1M) +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -1. Daftar: [MiniMax](https://www.minimax.io/) -2. Dapatkan kunci API -3. Dasbor → Tambahkan Kunci API + -**Gunakan:**`minimax/MiniMax-M2.1` +
+💰 Cheap Providers (Backup) -**Kiat Pro:**Opsi termurah untuk konteks panjang (1 juta token)!### Kimi K2 ($9/month flat) +### GLM-4.7 (Daily reset, $0.6/1M) -1. Berlangganan: [Moonshot AI](https://platform.moonshot.ai/) -2. Dapatkan kunci API -3. Dasbor → Tambahkan Kunci API +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**Gunakan:**`kimi/kimi-terbaru` +**Use:** `glm/glm-4.7` -**Kiat Pro:**Memperbaiki $9/bulan untuk 10 juta token = biaya efektif $0,90/1 juta!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. - -🆓 Penyedia GRATIS (Cadangan Darurat)### Qoder (5 FREE models via OAuth) +### MiniMax M2.1 (5h reset, $0.20/1M) + +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `minimax/MiniMax-M2.1` + +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1560,8 +1787,10 @@ Models:
- -🎨 Buat Kombo### Example 1: Maximize Subscription → Cheap Backup +
+🎨 Create Combos + +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1589,8 +1818,10 @@ Cost: $0 forever!
- -🔧 Integrasi CLI### Cursor IDE +
+🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1601,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Gunakan halaman**CLI Tools**di dasbor untuk konfigurasi sekali klik, atau edit `~/.claude/settings.json` secara manual.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1612,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Opsi 1 — Dasbor (disarankan):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Opsi 2 — Manual:**Edit `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1629,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Catatan:**OpenClaw hanya berfungsi dengan OmniRoute lokal. Gunakan `127.0.0.1` alih-alih `localhost` untuk menghindari masalah resolusi IPv6.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1643,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Langkah 1:**Tambahkan OmniRoute sebagai penyedia khusus:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Langkah 2:**Buat/edit `opencode.json` di root proyek Anda:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1669,117 +1909,130 @@ opencode } } } -```` +``` -**Langkah 3:**Pilih model di OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Tips:**Tambahkan model apa pun yang tersedia di titik akhir `/v1/models` OmniRoute Anda ke bagian `models`. Gunakan format `penyedia/model-id` dari dasbor OmniRoute Anda.
+ --- ## Pemecahan Masalah - -Klik untuk memperluas panduan pemecahan masalah +
+Click to expand troubleshooting guide -**"Model bahasa tidak memberikan pesan"** +**"Language model did not provide messages"** -- Kuota penyedia habis → Periksa dashboard pelacak kuota -- Solusi: Gunakan combo fallback atau beralih ke tier yang lebih murah +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**Pembatasan tarif** +**Rate limiting** -- Kuota berlangganan habis → Penggantian ke GLM/MiniMax -- Tambahkan kombo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**Token OAuth kedaluwarsa** +**OAuth token expired** -- Disegarkan secara otomatis oleh OmniRoute -- Jika masalah terus berlanjut: Dasbor → Penyedia → Sambungkan kembali +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Biaya tinggi** +**High costs** -- Periksa statistik penggunaan di Dashboard → Biaya -- Ganti model utama ke GLM/MiniMax -- Gunakan tingkat gratis (Gemini CLI, Qoder) untuk tugas-tugas yang tidak penting +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Port dasbor/API salah** +**Dashboard/API ports are wrong** -- `PORT` adalah port dasar kanonik (dan port API secara default) -- `API_PORT` hanya menimpa pendengar API yang kompatibel dengan OpenAI -- `DASHBOARD_PORT` hanya menimpa dashboard/pendengar Next.js -- Setel `NEXT_PUBLIC_BASE_URL` ke dasbor/URL publik Anda (untuk panggilan balik OAuth) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Kesalahan sinkronisasi cloud** +**Cloud sync errors** -- Verifikasikan `BASE_URL` menunjuk ke instance Anda yang sedang berjalan -- Verifikasikan `CLOUD_URL` menunjuk ke titik akhir cloud yang Anda harapkan -- Jaga agar nilai `NEXT_PUBLIC_*` tetap selaras dengan nilai sisi server +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Login pertama tidak berfungsi** +**First login not working** -- Centang `INITIAL_PASSWORD` di `.env` -- Jika tidak disetel, kata sandi cadangan adalah `123456` +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Tidak ada log permintaan** +**No request logs** -- Artefak permintaan ditulis ke `DATA_DIR/call_logs/` sebagai satu file JSON per permintaan -- Aktifkan pengambilan saluran pipa dari Dasbor → Log → Log Permintaan jika Anda memerlukan muatan per tahap yang terperinci -- Setel `APP_LOG_TO_FILE=true` jika Anda juga ingin log konsol aplikasi di `logs/application/app.log` -- Sesuaikan `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, dan `CALL_LOG_MAX_ENTRIES` sesuai kebutuhan +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Tes koneksi menunjukkan "Tidak valid" untuk penyedia yang kompatibel dengan OpenAI** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Banyak penyedia tidak mengekspos titik akhir `/models` -- OmniRoute v1.0.6+ menyertakan validasi fallback melalui penyelesaian obrolan -- Pastikan URL dasar menyertakan akhiran `/v1`### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix - +### 🔐 OAuth on a Remote Server + + ->**⚠️ Penting bagi pengguna yang menjalankan OmniRoute di VPS, Docker, atau server jarak jauh mana pun**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -Penyedia**Antigravitasi**dan**Gemini CLI**menggunakan**Google OAuth 2.0**. Google mewajibkan `redirect_uri` di alur OAuth agar sama persis dengan salah satu URI yang telah didaftarkan sebelumnya di Google Cloud Console aplikasi. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -Kredensial OAuth yang disertakan dalam OmniRoute didaftarkan**hanya untuk `localhost`**. Saat Anda mengakses OmniRoute di server jarak jauh (misalnya `https://omniroute.myserver.com`), Google menolak autentikasi dengan:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Anda perlu membuat**ID Klien OAuth 2.0**di Google Cloud Console dengan URI server Anda.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Buka Google Cloud Console** +#### Step-by-step -Buka: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Buat ID Klien OAuth 2.0 baru** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Klik**"+ Buat Kredensial"**→**"ID klien OAuth"** -- Jenis aplikasi:**"Aplikasi web"** -- Nama: apa pun yang Anda suka (misalnya `OmniRoute Remote`) +**2. Create a new OAuth 2.0 Client ID** -**3. Tambahkan URI Pengalihan Resmi** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -Di kolom**"URI pengalihan resmi"**, tambahkan:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Ganti `server-Anda.com` dengan domain atau IP server Anda (sertakan port jika diperlukan, misalnya `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. Simpan dan salin kredensial** +After creating, Google will show the **Client ID** and **Client Secret**. -Setelah pembuatan, Google akan menampilkan**ID Klien**dan**Rahasia Klien**. +**5. Set environment variables** -**5. Tetapkan variabel lingkungan** +In your `.env` (or Docker environment variables): -Di `.env` Anda (atau variabel lingkungan Docker):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1788,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Mulai ulang OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Coba sambungkan lagi** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Dasbor → Penyedia → Antigravitasi (atau Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google sekarang akan mengalihkan dengan benar ke `https://server-anda.com/callback`.--- +--- #### Temporary workaround (without custom credentials) -Jika Anda tidak ingin menyiapkan kredensial Anda sendiri saat ini, Anda masih dapat menggunakan**alur URL manual**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute membuka URL otorisasi Google -2. Setelah otorisasi, Google mencoba mengalihkan ke `localhost` (yang gagal di server jauh) -3.**Salin URL lengkap**dari bilah alamat browser Anda (meskipun halaman tidak dimuat) -4. Tempelkan URL tersebut ke bidang yang ditampilkan di modal koneksi OmniRoute -5. Klik**"Hubungkan"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Ini berfungsi karena kode otorisasi di URL valid terlepas dari apakah halaman pengalihan dimuat.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. - -🇧🇷 Versão em Português#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Pembuktiannya**Antigravitasi**dan**Gemini CLI**digunakan**Google OAuth 2.0**untuk autentikasi. Google meminta `redirect_uri` menggunakan OAuth yang terus berubah, jadi**exatamente**adalah URI yang sudah ada sebelumnya di Google Cloud Console yang dapat diterapkan. +
+🇧🇷 Versão em Português -Karena kredensial OAuth yang diberikan pada OmniRoute adalah kadastradas**apenas untuk `localhost`**. Ketika Anda mengakses OmniRoute dari server jarak jauh (misal: `https://omniroute.meuservidor.com`), atau Google meminta autenticação com:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Anda perlu membuat**ID Klien OAuth 2.0**di Google Cloud Console dengan URI di server Anda.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Mengakses Google Cloud Console** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -**2. Inilah ID Klien OAuth 2.0 yang baru** +**2. Crie um novo OAuth 2.0 Client ID** -- Klik pada**"+ Buat Kredensial"**→**"ID klien OAuth"** -- Tip aplikasi:**"Aplikasi web"** -- Nama: escolha qualquer nome (misal: `OmniRoute Remote`) +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Tambahan sebagai URI Pengalihan Resmi** +**3. Adicione as Authorized Redirect URIs** -Selain itu**"URI pengalihan resmi"**, tambahan:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Pengganti `seu-servidor.com` ke alamat IP atau alamat IP server Anda (termasuk port yang diperlukan, misal: `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. Salep dan salin sebagai kredensial** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Kemudian, Google menampilkan**ID Klien**dan**Rahasia Klien**. +**5. Configure as variáveis de ambiente** -**5. Konfigurasikan sebagai variáveis de ambiente** +No seu `.env` (ou nas variáveis de ambiente do Docker): -Tidak ada `.env` (atau berbagai lingkungan di Docker):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1867,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Memulai kembali OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute +``` -```` +**7. Tente conectar novamente** -**7. Mari kita sambungkan lagi** +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Dasbor → Penyedia → Antigravitasi (atau Gemini CLI) → OAuth +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. -Sekarang Google dialihkan ke `https://seu-servidor.com/callback` dan autentikasi berfungsi.--- +--- #### Workaround temporário (sem configurar credenciais próprias) -Jika Anda tidak ingin membuat kredensial pribadi sekarang, Anda mungkin dapat menggunakan fluks**panduan URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. O OmniRoute membuka URL autoriza dari Google -2. Setelah Anda autorizar, Google mengarahkan pengalihan ke `localhost` (yang gagal pada server jarak jauh) -3.**Salin URL lengkap**dari bilah akhir browser Anda (meskipun halaman tidak dimuat) -4. Gunakan URL ini karena tidak ada modal koneksi ke OmniRoute -5. Klik pada**"Hubungkan"** +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) +4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute +5. Clique em **"Connect"** -> Solusi ini berfungsi karena kode otorisasi pada URL valid secara independen untuk mengarahkan ulang ke akun atau tidak.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1905,64 +2171,73 @@ Jika Anda tidak ingin membuat kredensial pribadi sekarang, Anda mungkin dapat me ## 🛠️ Tech Stack - -Klik untuk memperluas detail tumpukan teknologi +
+Click to expand tech stack details --**Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+**tidak didukung**— biner asli `better-sqlite3` tidak kompatibel) --**Bahasa**: TypeScript 5.9 —**100% TypeScript**di `src/` dan `open-sse/` (tidak ada `any` di modul inti sejak v2.0) --**Kerangka Kerja**: Next.js 16 + React 19 + Tailwind CSS 4 --**Database**: LowDB (JSON) + SQLite (status domain + log proksi + audit MCP + keputusan perutean) --**Skema**: Zod (validasi I/O alat MCP, kontrak API) --**Protokol**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Streaming**: Peristiwa Terkirim Server (SSE) --**Auth**: OAuth 2.0 (PKCE) + JWT + Kunci API + Otorisasi Cakupan MCP --**Pengujian**: Pelari pengujian Node.js + Vitest (900+ pengujian termasuk unit, integrasi, E2E) --**CI/CD**: Tindakan GitHub (publikasi npm otomatis + Docker Hub saat rilis) --**Situs Web**: [omniroute.online](https://omniroute.online) --**Paket**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Ketahanan**: Pemutus arus, backoff eksponensial, kawanan anti-thundering, spoofing TLS, penyembuhan diri kombo otomatis
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Dokumentasi -| Dokumen | Deskripsi | +| Document | Description | | ---------------------------------------------- | --------------------------------------------------- | -| [Panduan Pengguna](docs/USER_GUIDE.md) | Penyedia, kombo, integrasi CLI, penerapan | -| [Referensi API](docs/API_REFERENCE.md) | Semua titik akhir dengan contoh | -| [Server MCP](open-sse/mcp-server/README.md) | 16 alat MCP, konfigurasi IDE, klien Python/TS/Go | -| [Server A2A](src/lib/a2a/README.md) | Protokol JSON-RPC 2.0, keterampilan, streaming, manajemen tugas | -| [Mesin Kombo Otomatis](docs/auto-combo.md) | Penilaian 6 faktor, paket mode, penyembuhan diri | -| [Pemecahan Masalah](docs/TROUBLESHOOTING.md) | Masalah umum dan solusinya | -| [Arsitektur](docs/ARCHITECTURE.md) | Arsitektur sistem dan internal | -| [Berkontribusi](BERKONTRIBUSI.md) | Pengaturan dan pedoman pengembangan | -| [Spesifikasi OpenAPI](docs/openapi.yaml) | Spesifikasi OpenAPI 3.0 | -| [Kebijakan Keamanan](SECURITY.md) | Pelaporan kerentanan dan praktik keamanan | -| [Penerapan VM](docs/VM_DEPLOYMENT_GUIDE.md) | Panduan lengkap: VM + nginx + Pengaturan Cloudflare | -| [Galeri Fitur](docs/FEATURES.md) | Tur dasbor visual dengan tangkapan layar | -| [Daftar Periksa Rilis](docs/RELEASE_CHECKLIST.md) | Langkah validasi pra-rilis |--- +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute memiliki**210+ fitur yang direncanakan**di berbagai fase pengembangan. Berikut adalah bidang-bidang utamanya: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Kategori | Fitur yang Direncanakan | Sorotan | -| ----------------------------- | ---------------- | ------------------------------------------------------------------------ | -| 🧠**Perutean & Kecerdasan**| 25+ | Perutean latensi terendah, perutean berbasis tag, preflight kuota, pemilihan akun P2C | -| 🔒**Keamanan & Kepatuhan**| 20+ | Pengerasan SSRF, penyelubungan kredensial, batas tarif per titik akhir, pelingkupan kunci manajemen | -| 📊**Kemampuan Observasi**| 15+ | Integrasi OpenTelemetry, pemantauan kuota waktu nyata, pelacakan biaya per model | -| 🔄**Integrasi Penyedia**| 20+ | Registri model dinamis, cooldown penyedia, Codex multi-akun, penguraian kuota Salinan | -| ⚡**Kinerja**| 15+ | Lapisan cache ganda, cache cepat, cache respons, streaming keepalive, API batch | -| 🌐**Ekosistem**| 10+ | WebSocket API, konfigurasi hot-reload, penyimpanan konfigurasi terdistribusi, mode komersial |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**Integrasi OpenCode**— Dukungan penyedia asli untuk IDE pengkodean AI OpenCode -- 🔗**Integrasi TRAE**— Dukungan penuh untuk kerangka pengembangan AI TRAE -- 📦**Batch API**— Pemrosesan batch asinkron untuk permintaan massal -- 🎯**Perutean Berbasis Tag**— Merutekan permintaan berdasarkan tag dan metadata khusus -- 💰**Strategi Biaya Terendah**— Secara otomatis memilih penyedia termurah yang tersedia +### 🔜 Coming Soon -> 📝 Spesifikasi fitur lengkap tersedia di [`docs/new-features/`](docs/new-features/) (217 spesifikasi detail)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1970,18 +2245,20 @@ OmniRoute memiliki**210+ fitur yang direncanakan**di berbagai fase pengembangan. ### How to Contribute -1. Cabangkan repositori -2. Buat cabang fitur Anda (`git checkout -b feature/amazing-feature`) -3. Komit perubahan Anda (`git commit -m 'Tambahkan fitur luar biasa'`) -4. Dorong ke cabang (`fitur asal git push/fitur luar biasa`) -5. Buka Permintaan Tarik +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Lihat [CONTRIBUTING.md](CONTRIBUTING.md) untuk pedoman rinci.### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1993,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Terima kasih khusus kepada**[9router](https://github.com/decolua/9router)**oleh**[decolua](https://github.com/decolua)**— proyek asli yang menginspirasi fork ini. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Terima kasih khusus kepada**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— implementasi Go asli yang menginspirasi port JavaScript ini.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Lisensi -Lisensi MIT - lihat [LISENSI](LISENSI) untuk detailnya.--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/id/docs/ARCHITECTURE.md b/docs/i18n/id/docs/ARCHITECTURE.md index 8e302a823f..99420b9f46 100644 --- a/docs/i18n/id/docs/ARCHITECTURE.md +++ b/docs/i18n/id/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Terakhir diperbarui: 28-03-2026_## Executive Summary -OmniRoute adalah gateway dan dasbor perutean AI lokal yang dibangun di Next.js. -Ini menyediakan satu titik akhir yang kompatibel dengan OpenAI (`/v1/*`) dan merutekan lalu lintas di beberapa penyedia upstream dengan terjemahan, fallback, penyegaran token, dan pelacakan penggunaan. -Kemampuan inti: +_Last updated: 2026-03-28_ -- Permukaan API yang kompatibel dengan OpenAI untuk CLI/alat (28 penyedia) -- Permintaan/tanggapan terjemahan lintas format penyedia -- Model kombo fallback (urutan multi-model) -- Penggantian tingkat akun (multi-akun per penyedia) -- Manajemen koneksi penyedia kunci OAuth + API -- Menyematkan generasi melalui `/v1/embeddings` (6 penyedia, 9 model) -- Pembuatan gambar melalui `/v1/images/generasi` (4 penyedia, 9 model) -- Penguraian tag Think (`...`) untuk model penalaran -- Sanitasi respons untuk kompatibilitas OpenAI SDK yang ketat -- Normalisasi peran (pengembang→sistem, sistem→pengguna) untuk kompatibilitas lintas penyedia -- Konversi keluaran terstruktur (json_schema → Gemini responSchema) -- Persistensi lokal untuk penyedia, kunci, alias, kombo, pengaturan, harga -- Pelacakan penggunaan/biaya dan pencatatan permintaan -- Sinkronisasi cloud opsional untuk sinkronisasi multi-perangkat/negara -- Daftar IP yang diizinkan/daftar blokir untuk kontrol akses API -- Memikirkan manajemen anggaran (passthrough/otomatis/custom/adaptif) -- Injeksi cepat sistem global -- Pelacakan sesi dan sidik jari -- Pembatasan tarif yang ditingkatkan per akun dengan profil khusus penyedia -- Pola pemutus sirkuit untuk ketahanan penyedia -- Perlindungan kawanan anti guntur dengan penguncian mutex -- Cache deduplikasi permintaan berbasis tanda tangan -- Lapisan domain: ketersediaan model, aturan biaya, kebijakan fallback, kebijakan lockout -- Persistensi status domain (cache tulis SQLite untuk fallback, anggaran, penguncian, pemutus sirkuit) -- Mesin kebijakan untuk evaluasi permintaan terpusat (lockout → anggaran → fallback) -- Minta telemetri dengan agregasi latensi p50/p95/p99 -- ID Korelasi (X-Request-Id) untuk penelusuran ujung ke ujung -- Pencatatan audit kepatuhan dengan opt-out per kunci API -- Kerangka evaluasi untuk penjaminan mutu LLM -- Dasbor UI ketahanan dengan status pemutus sirkuit waktu nyata -- Penyedia OAuth modular (12 modul individual di bawah `src/lib/oauth/providers/`) +## Executive Summary -Model waktu proses utama: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- Rute aplikasi Next.js di bawah `src/app/api/*` mengimplementasikan API dasbor dan API kompatibilitas -- Inti SSE/perutean bersama di `src/sse/*` + `open-sse/*` menangani eksekusi penyedia, terjemahan, streaming, fallback, dan penggunaan## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Waktu aktif gateway lokal -- API manajemen dasbor -- Otentikasi penyedia dan penyegaran token -- Minta terjemahan dan streaming SSE -- Status lokal + persistensi penggunaan -- Orkestrasi sinkronisasi cloud opsional### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Implementasi layanan cloud di belakang `NEXT_PUBLIC_CLOUD_URL` -- Penyedia SLA/bidang kontrol di luar proses lokal -- Biner CLI eksternal itu sendiri (Claude CLI, Codex CLI, dll.)## Dashboard Surface (Current) +### Out of Scope -Halaman utama di bawah `src/app/(dashboard)/dashboard/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dasbor` — mulai cepat + ikhtisar penyedia -- `/dashboard/endpoint` — proksi titik akhir + MCP + A2A + tab titik akhir API -- `/dashboard/providers` — koneksi dan kredensial penyedia -- `/dashboard/combos` — strategi kombo, templat, aturan perutean model -- `/dashboard/cost` — agregasi biaya dan visibilitas harga -- `/dashboard/analytics` — analisis dan evaluasi penggunaan -- `/dasbor/batas` — kontrol kuota/tarif -- `/dashboard/cli-tools` — Orientasi CLI, deteksi runtime, pembuatan konfigurasi -- `/dashboard/agents` — mendeteksi agen ACP + pendaftaran agen khusus -- `/dasbor/media` — taman bermain gambar/video/musik -- `/dashboard/search-tools` — pengujian dan riwayat penyedia pencarian -- `/dasbor/kesehatan` — waktu aktif, pemutus sirkuit, batas kecepatan -- `/dashboard/logs` — log permintaan/proksi/audit/konsol -- `/dashboard/settings` — tab pengaturan sistem (umum, perutean, default kombo, dll.) -- `/dashboard/api-manager` — siklus hidup kunci API dan izin model## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Direktori utama: +Main directories: -- `src/app/api/v1/*` dan `src/app/api/v1beta/*` untuk API kompatibilitas -- `src/app/api/*` untuk API manajemen/konfigurasi -- Selanjutnya penulisan ulang di `next.config.mjs` peta `/v1/*` menjadi `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Rute kompatibilitas penting: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` — menyertakan model khusus dengan `custom: true` -- `src/app/api/v1/embeddings/route.ts` — pembuatan penyematan (6 penyedia) -- `src/app/api/v1/images/generasi/route.ts` — pembuatan gambar (4+ penyedia termasuk Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — obrolan khusus per penyedia -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — penyematan khusus per penyedia -- `src/app/api/v1/providers/[provider]/images/generasi/route.ts` — gambar khusus per penyedia +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` - `src/app/api/v1beta/models/[...path]/route.ts` -Domain manajemen: +Management domains: -- Otentikasi/pengaturan: `src/app/api/auth/*`, `src/app/api/settings/*` -- Penyedia/koneksi: `src/app/api/providers*` -- Node penyedia: `src/app/api/provider-nodes*` -- Model khusus: `src/app/api/provider-models` (GET/POST/DELETE) -- Katalog model: `src/app/api/models/route.ts` (GET) -- Konfigurasi proxy: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- Kunci/alias/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- Penggunaan: `src/app/api/usage/*` -- Sinkronisasi/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` -- Pembantu perkakas CLI: `src/app/api/cli-tools/*` -- Filter IP: `src/app/api/settings/ip-filter` (GET/PUT) -- Memikirkan anggaran: `src/app/api/settings/thinking-budget` (GET/PUT) -- Perintah sistem: `src/app/api/settings/system-prompt` (GET/PUT) -- Sesi: `src/app/api/sessions` (GET) -- Batas tarif: `src/app/api/rate-limits` (GET) -- Ketahanan: `src/app/api/resilience` (GET/PATCH) — profil penyedia, pemutus sirkuit, status batas kecepatan -- Reset ketahanan: `src/app/api/resilience/reset` (POST) — reset pemutus + cooldown -- Statistik cache: `src/app/api/cache/stats` (DAPATKAN/HAPUS) -- Ketersediaan model: `src/app/api/models/availability` (GET/POST) -- Telemetri: `src/app/api/telemetri/ringkasan` (GET) -- Anggaran: `src/app/api/usage/budget` (GET/POST) -- Rantai cadangan: `src/app/api/fallback/chains` (GET/POST/DELETE) -- Audit kepatuhan: `src/app/api/compliance/audit-log` (GET) -- Evaluasi: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- Kebijakan: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -Modul aliran utama: +## 2) SSE + Translation Core -- Entri: `src/sse/handlers/chat.ts` -- Orkestrasi inti: `open-sse/handlers/chatCore.ts` -- Adaptor eksekusi penyedia: `open-sse/executors/*` -- Deteksi format/konfigurasi penyedia: `open-sse/services/provider.ts` -- Penguraian/penyelesaian model: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Logika penggantian akun: `open-sse/services/accountFallback.ts` -- Registri terjemahan: `open-sse/translator/index.ts` -- Transformasi aliran: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- Ekstraksi/normalisasi penggunaan: `open-sse/utils/usageTracking.ts` -- Pikirkan pengurai tag: `open-sse/utils/thinkTagParser.ts` -- Pengendali penyematan: `open-sse/handlers/embeddings.ts` -- Menyematkan registri penyedia: `open-sse/config/embeddingRegistry.ts` -- Pengendali pembuatan gambar: `open-sse/handlers/imageGeneration.ts` -- Registri penyedia gambar: `open-sse/config/imageRegistry.ts` -- Sanitasi respons: `open-sse/handlers/responseSanitizer.ts` -- Normalisasi peran: `open-sse/services/roleNormalizer.ts` +Main flow modules: -Layanan (logika bisnis): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- Pemilihan/penilaian akun: `open-sse/services/accountSelector.ts` -- Manajemen siklus hidup konteks: `open-sse/services/contextManager.ts` -- Penegakan filter IP: `open-sse/services/ipFilter.ts` -- Pelacakan sesi: `open-sse/services/sessionManager.ts` -- Minta deduplikasi: `open-sse/services/signatureCache.ts` -- Injeksi cepat sistem: `open-sse/services/systemPrompt.ts` -- Berpikir manajemen anggaran: `open-sse/services/thinkingBudget.ts` -- Perutean model wildcard: `open-sse/services/wildcardRouter.ts` -- Manajemen batas tarif: `open-sse/services/rateLimitManager.ts` -- Pemutus sirkuit: `open-sse/services/circirBreaker.ts` +Services (business logic): -Modul lapisan domain: +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- Ketersediaan model: `src/lib/domain/modelAvailability.ts` -- Aturan/anggaran biaya: `src/lib/domain/costRules.ts` -- Kebijakan cadangan: `src/lib/domain/fallbackPolicy.ts` -- Penyelesai kombo: `src/lib/domain/comboResolver.ts` -- Kebijakan penguncian: `src/lib/domain/lockoutPolicy.ts` -- Mesin kebijakan: `src/domain/policyEngine.ts` — penguncian terpusat → anggaran → evaluasi cadangan -- Katalog kode kesalahan: `src/lib/domain/errorCodes.ts` -- ID Permintaan: `src/lib/domain/requestId.ts` -- Batas waktu pengambilan: `src/lib/domain/fetchTimeout.ts` -- Permintaan telemetri: `src/lib/domain/requestTelemetry.ts` -- Kepatuhan/audit: `src/lib/domain/compliance/index.ts` -- Pelari evaluasi: `src/lib/domain/evalRunner.ts` -- Persistensi status domain: `src/lib/db/domainState.ts` — SQLite CRUD untuk rantai fallback, anggaran, riwayat biaya, status lockout, pemutus sirkuit +Domain layer modules: -Modul penyedia OAuth (12 file individual di bawah `src/lib/oauth/providers/`): +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -- Indeks registri: `src/lib/oauth/providers/index.ts` -- Penyedia individu: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` -- Pembungkus tipis: `src/lib/oauth/providers.ts` — mengekspor ulang dari masing-masing modul## 3) Persistence Layer +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -DB status utama (SQLite): +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -- Infra inti: `src/lib/db/core.ts` (lebih baik-sqlite3, migrasi, WAL) -- Ekspor ulang fasad: `src/lib/localDb.ts` (lapisan kompatibilitas tipis untuk penelepon) -- file: `${DATA_DIR}/storage.sqlite` (atau `$XDG_CONFIG_HOME/omniroute/storage.sqlite` bila disetel, jika tidak `~/.omniroute/storage.sqlite`) -- entitas (tabel + namespace KV): ProviderConnections, ProviderNodes, ModelAliases, Combo, ApiKeys, Pengaturan, Harga,**customModels**,**proxyConfig**,**ipFilter**,**ThinkingBudget**,**systemPrompt** +## 3) Persistence Layer -Kegigihan penggunaan: +Primary state DB (SQLite): -- fasad: `src/lib/usageDb.ts` (modul yang didekomposisi di `src/lib/usage/*`) -- Tabel SQLite di `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` -- artefak file opsional tetap ada untuk kompatibilitas/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- File JSON lama dimigrasikan ke SQLite melalui migrasi startup jika ada +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -DB Status Domain (SQLite): +Usage persistence: -- `src/lib/db/domainState.ts` — Operasi CRUD untuk status domain -- Tabel (dibuat di `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circir_breakers` -- Pola cache write-through: Peta dalam memori bersifat otoritatif saat runtime; mutasi ditulis secara sinkron ke SQLite; keadaan dipulihkan dari DB pada start dingin## 4) Auth + Security Surfaces +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- Otentikasi cookie dasbor: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- Pembuatan/verifikasi kunci API: `src/shared/utils/apiKey.ts` -- Rahasia penyedia tetap ada di entri `providerConnections` -- Dukungan proxy keluar melalui `open-sse/utils/proxyFetch.ts` (env vars) dan `open-sse/utils/networkProxy.ts` (dapat dikonfigurasi per penyedia atau global)## 5) Cloud Sync +Domain State DB (SQLite): -- Penjadwal init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Tugas berkala: `src/shared/services/cloudSyncScheduler.ts` -- Tugas berkala: `src/shared/services/modelSyncScheduler.ts` -- Rute kontrol: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Keputusan fallback didorong oleh `open-sse/services/accountFallback.ts` menggunakan kode status dan heuristik pesan kesalahan. Perutean kombo menambahkan satu perlindungan tambahan: 400 dengan cakupan penyedia seperti blok konten upstream dan kegagalan validasi peran diperlakukan sebagai kegagalan model lokal sehingga target kombo selanjutnya masih dapat berjalan.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -Penyegaran selama lalu lintas langsung dijalankan di dalam `open-sse/handlers/chatCore.ts` melalui pelaksana `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -Sinkronisasi berkala dipicu oleh `CloudSyncScheduler` saat cloud diaktifkan.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -File penyimpanan fisik: +Physical storage files: -- DB waktu proses utama: `${DATA_DIR}/storage.sqlite` -- baris log permintaan: `${DATA_DIR}/log.txt` (artefak compat/debug) -- arsip muatan panggilan terstruktur: `${DATA_DIR}/call_logs/` -- sesi debug penerjemah/permintaan opsional: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: API kompatibilitas -- `src/app/api/v1/providers/[provider]/*`: rute khusus per penyedia (obrolan, penyematan, gambar) -- `src/app/api/providers*`: penyedia CRUD, validasi, pengujian -- `src/app/api/provider-nodes*`: manajemen node khusus yang kompatibel -- `src/app/api/provider-models`: manajemen model khusus (CRUD) -- `src/app/api/models/route.ts`: API katalog model (alias + model khusus) -- `src/app/api/oauth/*`: OAuth/kode perangkat mengalir -- `src/app/api/keys*`: siklus hidup kunci API lokal -- `src/app/api/models/alias`: manajemen alias -- `src/app/api/combos*`: manajemen kombo cadangan -- `src/app/api/pricing`: penggantian harga untuk penghitungan biaya -- `src/app/api/settings/proxy`: konfigurasi proxy (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: uji konektivitas proxy keluar (POST) -- `src/app/api/usage/*`: penggunaan dan log API -- `src/app/api/sync/*` + `src/app/api/cloud/*`: sinkronisasi cloud dan bantuan yang menghadap cloud -- `src/app/api/cli-tools/*`: penulis/pemeriksa konfigurasi CLI lokal -- `src/app/api/settings/ip-filter`: Daftar IP yang diizinkan/daftar blokir (GET/PUT) -- `src/app/api/settings/thinking-budget`: konfigurasi anggaran token pemikiran (GET/PUT) -- `src/app/api/settings/system-prompt`: perintah sistem global (GET/PUT) -- `src/app/api/sessions`: daftar sesi aktif (GET) -- `src/app/api/rate-limits`: status batas tarif per akun (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: penguraian permintaan, penanganan kombo, putaran pemilihan akun -- `open-sse/handlers/chatCore.ts`: terjemahan, pengiriman eksekutor, coba lagi/penyegaran penanganan, pengaturan streaming -- `open-sse/executors/*`: perilaku format dan jaringan khusus penyedia### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: registrasi dan orkestrasi penerjemah -- Permintaan penerjemah: `open-sse/translator/request/*` -- Penerjemah respons: `open-sse/translator/response/*` -- Konstanta format: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: konfigurasi/status persisten dan persistensi domain di SQLite -- `src/lib/localDb.ts`: ekspor ulang kompatibilitas untuk modul DB -- `src/lib/usageDb.ts`: riwayat penggunaan/log panggilan fasad di atas tabel SQLite## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Setiap penyedia memiliki pelaksana khusus yang memperluas `BaseExecutor` (dalam `open-sse/executors/base.ts`), yang menyediakan pembuatan URL, konstruksi header, percobaan ulang dengan backoff eksponensial, kait penyegaran kredensial, dan metode orkestrasi `execute()`. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Pelaksana | Penyedia | Penanganan Khusus | -| ------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------- | -| `Eksekutor Default` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Kebingungan, Bersama-sama, Kembang Api, Cerebras, Cohere, NVIDIA | Konfigurasi URL/tajuk dinamis per penyedia | -| `Pelaksana Antigravitasi` | Google Antigravitasi | ID proyek/sesi khusus, Coba Lagi-Setelah penguraian | -| `Pelaksana Codex` | Kodeks OpenAI | Menyuntikkan instruksi sistem, memaksakan upaya penalaran | -| `Pelaksana Kursor` | IDE Kursor | Protokol ConnectRPC, pengkodean Protobuf, penandatanganan permintaan melalui checksum | -| `GithubExecutor` | Kopilot GitHub | Penyegaran token kopilot, header yang meniru VSCode | -| `Pelaksana Kiro` | AWS CodeWhisperer/Kiro | Format biner AWS EventStream → konversi SSE | -| `Eksekutor GeminiCLI` | CLI Gemini | Siklus penyegaran token Google OAuth | +### Persistence -Semua penyedia lain (termasuk node khusus yang kompatibel) menggunakan `DefaultExecutor`.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Penyedia | Format | Otentikasi | Aliran | Non-Aliran | Penyegaran Token | API Penggunaan | -| ---------------- | ---------------- | --------------------- | ----------------- | ---------- | ---------------- | --------------------- | ------------------------------ | -| Claude | claude | Kunci API / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin saja | -| kembar | gemilang | Kunci API / OAuth | ✅ | ✅ | ✅ | ⚠️ Konsol Cloud | -| CLI Gemini | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Konsol Cloud | -| Antigravitasi | antigravitasi | OAuth | ✅ | ✅ | ✅ | ✅ API kuota penuh | -| OpenAI | buka | Kunci API | ✅ | ✅ | ❌ | ❌ | -| Kodeks | openai-tanggapan | OAuth | ✅ dipaksa | ❌ | ✅ | ✅ Batas tarif | -| Kopilot GitHub | buka | OAuth + Token Kopilot | ✅ | ✅ | ✅ | ✅ Cuplikan kuota | -| Kursor | kursor | Checksum khusus | ✅ | ✅ | ❌ | ❌ | -| Kiro | kiri | AWSSSO OIDC | ✅ (Aliran Acara) | ❌ | ✅ | ✅ Batasan penggunaan | -| Qwen | buka | OAuth | ✅ | ✅ | ✅ | ⚠️ Sesuai permintaan | -| Qoder | buka | OAuth (Dasar) | ✅ | ✅ | ✅ | ⚠️ Sesuai permintaan | -| BukaRouter | buka | Kunci API | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | claude | Kunci API | ✅ | ✅ | ❌ | ❌ | -| Pencarian Dalam | buka | Kunci API | ✅ | ✅ | ❌ | ❌ | -| Bagus | buka | Kunci API | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | buka | Kunci API | ✅ | ✅ | ❌ | ❌ | -| Mistral | buka | Kunci API | ✅ | ✅ | ❌ | ❌ | -| Kebingungan | buka | Kunci API | ✅ | ✅ | ❌ | ❌ | -| Bersama AI | buka | Kunci API | ✅ | ✅ | ❌ | ❌ | -| AI kembang api | buka | Kunci API | ✅ | ✅ | ❌ | ❌ | -| Otak | buka | Kunci API | ✅ | ✅ | ❌ | ❌ | -| menyatu | buka | Kunci API | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | buka | Kunci API | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Format sumber yang terdeteksi meliputi: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. -- `buka` -- `openai-tanggapan` +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | + +All other providers (including custom compatible nodes) use the `DefaultExecutor`. + +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: + +- `openai` +- `openai-responses` - `claude` - `gemini` -Format sasaran meliputi: +Target formats include: -- Obrolan/Respon OpenAI +- OpenAI chat/Responses - Claude -- Amplop Gemini/Gemini-CLI/Antigravitasi +- Gemini/Gemini-CLI/Antigravity envelope - Kiro -- Kursor +- Cursor -Penerjemahan menggunakan**OpenAI sebagai format hub**— semua konversi melalui OpenAI sebagai perantara:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Terjemahan dipilih secara dinamis berdasarkan bentuk muatan sumber dan format target penyedia. +Additional processing layers in the translation pipeline: -Lapisan pemrosesan tambahan dalam alur terjemahan: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Sanitasi respons**— Menghapus kolom non-standar dari respons format OpenAI (streaming dan non-streaming) untuk memastikan kepatuhan SDK yang ketat --**Normalisasi peran**— Mengonversi `developer` → `system` untuk target non-OpenAI; menggabungkan `sistem` → `pengguna` untuk model yang menolak peran sistem (GLM, ERNIE) --**Think tag ekstraksi**— Mengurai `...` blok dari konten ke dalam kolom `reasoning_content` --**Output terstruktur**— Mengonversi `response_format.json_schema` OpenAI menjadi `responseMimeType` + `responseSchema` Gemini## Supported API Endpoints +## Supported API Endpoints -| Titik akhir | Format | Penangan | -| --------------------------------------------------- | ---- | ------------------------------------------------------------------- | -| `POST /v1/obrolan/penyelesaian` | Obrolan OpenAI | `src/sse/handlers/chat.ts` | -| `POSTING /v1/pesan` | Pesan Claude | Penangan yang sama (terdeteksi otomatis) | -| `POSTING /v1/tanggapan` | Tanggapan OpenAI | `open-sse/handlers/responsesHandler.ts` | -| `POSTING /v1/embeddings` | Penyematan OpenAI | `open-sse/handlers/embeddings.ts` | -| `DAPATKAN /v1/embeddings` | Daftar model | Rute API | -| `POST /v1/gambar/generasi` | Gambar OpenAI | `open-sse/handlers/imageGeneration.ts` | -| `DAPATKAN /v1/gambar/generasi` | Daftar model | Rute API | -| `POST /v1/providers/{provider}/chat/completions` | Obrolan OpenAI | Per penyedia khusus dengan validasi model | -| `POST /v1/providers/{provider}/embeddings` | Penyematan OpenAI | Per penyedia khusus dengan validasi model | -| `POST /v1/providers/{provider}/images/generasi` | Gambar OpenAI | Per penyedia khusus dengan validasi model | -| `POST /v1/messages/count_tokens` | Jumlah Token Claude | Rute API | -| `DAPATKAN /v1/model` | Daftar Model OpenAI | Rute API (obrolan + penyematan + gambar + model khusus) | -| `DAPATKAN /api/model/katalog` | Katalog | Semua model dikelompokkan berdasarkan penyedia + tipe | -| `POST /v1beta/models/*:streamGenerateContent` | Gemini asli | Rute API | -| `GET/PUT/HAPUS /api/settings/proxy` | Konfigurasi Proksi | Konfigurasi proksi jaringan | -| `POST /api/settings/proxy/test` | Konektivitas Proksi | Titik akhir pengujian kesehatan/konektivitas proxy | -| `GET/POST/HAPUS /api/provider-models` | Model Penyedia | Metadata model penyedia mendukung model kustom dan terkelola yang tersedia |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -Penangan bypass (`open-sse/utils/bypassHandler.ts`) mencegat permintaan "sekali pakai" yang diketahui dari Claude CLI — ping pemanasan, ekstraksi judul, dan jumlah token — dan mengembalikan**respons palsu**tanpa menggunakan token penyedia upstream. Ini dipicu hanya ketika `Agen-Pengguna` berisi `claude-cli`.## Request Logger Pipeline +## Bypass Handler -Pencatat permintaan (`open-sse/utils/requestLogger.ts`) menyediakan pipeline logging debug 7 tahap, dinonaktifkan secara default, diaktifkan melalui `ENABLE_REQUEST_LOGS=true`:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -File ditulis ke `/logs//` untuk setiap sesi permintaan.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- cooldown akun penyedia pada kesalahan sementara/rate/auth -- penggantian akun sebelum permintaan gagal -- penggantian model kombo ketika jalur model/penyedia saat ini habis## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- pra-periksa dan segarkan dengan coba lagi untuk penyedia yang dapat disegarkan -- 401/403 percobaan ulang setelah upaya penyegaran di jalur inti## 3) Stream Safety +## 2) Token Expiry -- pengontrol aliran yang sadar akan pemutusan hubungan -- aliran terjemahan dengan flush akhir aliran dan penanganan `[SELESAI]` -- penggantian estimasi penggunaan ketika metadata penggunaan penyedia tidak ada## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- kesalahan sinkronisasi muncul tetapi runtime lokal terus berlanjut -- penjadwal memiliki logika yang mampu mencoba ulang, namun eksekusi berkala saat ini memanggil sinkronisasi upaya tunggal secara default## 5) Data Integrity +## 3) Stream Safety -- Migrasi skema SQLite dan kait pemutakhiran otomatis saat startup -- JSON lama → jalur kompatibilitas migrasi SQLite## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Sumber visibilitas waktu proses: +## 4) Cloud Sync Degradation -- log konsol dari `src/sse/utils/logger.ts` -- agregat penggunaan per permintaan di SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- pengambilan muatan terperinci empat tahap dalam SQLite (`request_detail_logs`) ketika `settings.detailed_logs_enabled=true` -- status permintaan tekstual masuk `log.txt` (opsional/compat) -- log permintaan/terjemahan dalam opsional di bawah `logs/` ketika `ENABLE_REQUEST_LOGS=true` -- titik akhir penggunaan dasbor (`/api/usage/*`) untuk konsumsi UI +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -Penangkapan payload permintaan terperinci menyimpan hingga empat tahap payload JSON per panggilan yang dirutekan: +## 5) Data Integrity -- permintaan mentah diterima dari klien -- permintaan yang diterjemahkan sebenarnya dikirim ke hulu -- respons penyedia direkonstruksi sebagai JSON; tanggapan yang dialirkan dipadatkan ke ringkasan akhir ditambah metadata aliran -- respons klien akhir yang dikembalikan oleh OmniRoute; tanggapan yang dialirkan disimpan dalam bentuk ringkasan ringkas yang sama## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- Rahasia JWT (`JWT_SECRET`) mengamankan verifikasi/penandatanganan cookie sesi dasbor -- Bootstrap kata sandi awal (`INITIAL_PASSWORD`) harus dikonfigurasi secara eksplisit untuk provisi yang dijalankan pertama kali -- Rahasia kunci API HMAC (`API_KEY_SECRET`) mengamankan format kunci API lokal yang dihasilkan -- Rahasia penyedia (kunci/token API) disimpan di DB lokal dan harus dilindungi di tingkat sistem file -- Titik akhir sinkronisasi cloud mengandalkan autentikasi kunci API + semantik id mesin## Environment and Runtime Matrix +## Observability and Operational Signals -Variabel lingkungan yang aktif digunakan oleh kode: +Runtime visibility sources: -- Aplikasi/autentikasi: `JWT_SECRET`, `INITIAL_PASSWORD` -- Penyimpanan: `DATA_DIR` -- Perilaku node yang kompatibel: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Penggantian basis penyimpanan opsional (Linux/macOS ketika `DATA_DIR` tidak disetel): `XDG_CONFIG_HOME` -- Hashing keamanan: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Pencatatan: `ENABLE_REQUEST_LOGS` -- URL sinkronisasi/cloud: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- Proksi keluar: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` dan varian huruf kecil -- Tanda fitur SOCKS5: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- Pembantu platform/runtime (bukan konfigurasi khusus aplikasi): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` dan `localDb` berbagi kebijakan direktori dasar yang sama (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) dengan migrasi file lama. -2. `/api/v1/route.ts` mendelegasikan ke pembuat katalog terpadu yang sama dengan yang digunakan oleh `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) untuk menghindari penyimpangan semantik. -3. Pencatat permintaan menulis header/isi lengkap saat diaktifkan; memperlakukan direktori log sebagai sensitif. -4. Perilaku cloud bergantung pada `NEXT_PUBLIC_BASE_URL` yang benar dan jangkauan titik akhir cloud. -5. Direktori `open-sse/` diterbitkan sebagai `@omniroute/open-sse`**paket ruang kerja npm**. Kode sumber mengimpornya melalui `@omniroute/open-sse/...` (diselesaikan dengan `transpilePackages` Next.js). Jalur file dalam dokumen ini masih menggunakan nama direktori `open-sse/` untuk konsistensi. -6. Bagan di dasbor menggunakan**Recharts**(berbasis SVG) untuk visualisasi analitik interaktif yang mudah diakses (diagram batang penggunaan model, tabel perincian penyedia dengan tingkat keberhasilan). -7. Tes E2E menggunakan**Playwright**(`tests/e2e/`), dijalankan melalui `npm run test:e2e`. Pengujian unit menggunakan**Node.js test runner**(`tests/unit/`), dijalankan melalui `npm run test:unit`. Kode sumber di bawah `src/` adalah**TypeScript**(`.ts`/`.tsx`); ruang kerja `open-sse/` tetap berupa JavaScript (`.js`). -8. Halaman pengaturan disusun dalam 5 tab: Keamanan, Perutean (6 strategi global: isi dulu, round-robin, p2c, acak, jarang digunakan, optimal biaya), Ketahanan (batas kecepatan yang dapat diedit, pemutus sirkuit, kebijakan), AI (anggaran berpikir, perintah sistem, cache cepat), Lanjutan (proxy).## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- Bangun dari sumber: `npm run build` -- Bangun gambar Docker: `docker build -t omniroute .` -- Mulai layanan dan verifikasi: -- `DAPATKAN /api/pengaturan` -- `DAPATKAN /api/v1/model` -- URL basis target CLI harus `http://:20128/v1` ketika `PORT=20128` +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` +- `GET /api/v1/models` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/id/docs/FEATURES.md b/docs/i18n/id/docs/FEATURES.md index ed1be08a13..9b86ac264f 100644 --- a/docs/i18n/id/docs/FEATURES.md +++ b/docs/i18n/id/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Panduan visual untuk setiap bagian dasbor OmniRoute.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Kelola koneksi penyedia AI: Penyedia OAuth (Claude Code, Codex, Gemini CLI), penyedia kunci API (Groq, DeepSeek, OpenRouter), dan penyedia gratis (Qoder, Qwen, Kiro). Akun Kiro mencakup pelacakan saldo kredit — sisa kredit, total penyisihan, dan tanggal perpanjangan yang terlihat di Dasbor → Penggunaan.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Buat kombo perutean model dengan 6 strategi: prioritas, berbobot, round-robin, acak, paling jarang digunakan, dan hemat biaya. Setiap kombo merangkai beberapa model dengan fallback otomatis dan mencakup templat cepat serta pemeriksaan kesiapan.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Analisis penggunaan yang komprehensif dengan konsumsi token, perkiraan biaya, peta panas aktivitas, grafik distribusi mingguan, dan perincian per penyedia.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Pemantauan waktu nyata: waktu aktif, memori, versi, persentil latensi (p50/p95/p99), statistik cache, dan status pemutus sirkuit penyedia.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Empat mode untuk men-debug terjemahan API:**Playground**(konverter format),**Chat Tester**(permintaan langsung),**Test Bench**(pengujian batch), dan**Live Monitor**(streaming real-time).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Uji model apa pun langsung dari dasbor. Pilih penyedia, model, dan titik akhir, tulis perintah dengan Monaco Editor, streaming respons secara real-time, batalkan mid-stream, dan lihat metrik waktu.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Tema warna yang dapat disesuaikan untuk seluruh dasbor. Pilih dari 7 warna preset (Coral, Blue, Red, Green, Violet, Orange, Cyan) atau buat tema khusus dengan memilih warna hex apa pun. Mendukung mode terang, gelap, dan sistem.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Panel pengaturan komprehensif dengan tab: +Comprehensive settings panel with tabs: --**Umum**— Penyimpanan sistem, manajemen cadangan (database ekspor/impor) -**Penampilan**— Pemilih tema (gelap/terang/sistem), preset tema warna dan warna khusus, visibilitas log kesehatan, kontrol visibilitas item sidebar -**Keamanan**— Perlindungan titik akhir API, pemblokiran penyedia khusus, pemfilteran IP, info sesi -**Perutean**— Alias model, degradasi tugas latar belakang -**Ketahanan**— Persistensi batas nilai, penyetelan pemutus sirkuit, penonaktifan otomatis akun yang diblokir, pemantauan kedaluwarsa penyedia -**Lanjutan**— Penggantian konfigurasi, jejak audit konfigurasi, mode degradasi fallback![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Konfigurasi sekali klik untuk alat pengkodean AI: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, dan Factory Droid. Menampilkan penerapan/reset konfigurasi otomatis, profil koneksi, dan pemetaan model.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Dasbor untuk menemukan dan mengelola agen CLI. Menampilkan kisi 14 agen bawaan (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) dengan: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Status instalasi**— Terpasang / Tidak Ditemukan dengan deteksi versi -**Lencana protokol**— stdio, HTTP, dll. -**Agen khusus**— Daftarkan alat CLI apa pun melalui formulir (nama, biner, perintah versi, argumen spawn) -**Pencocokan Sidik Jari CLI**— Tombol per penyedia untuk mencocokkan tanda tangan permintaan CLI asli, mengurangi risiko larangan sekaligus mempertahankan IP proxy--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Generate images, videos, and music from the dashboard. Mendukung OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, dan MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Pencatatan permintaan secara real-time dengan pemfilteran berdasarkan penyedia, model, akun, dan kunci API. Menampilkan kode status, penggunaan token, latensi, dan detail respons.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Titik akhir API terpadu Anda dengan perincian kemampuan: Penyelesaian Obrolan, API Respons, Penyematan, Pembuatan Gambar, Pemeringkatan Ulang, Transkripsi Audio, Text-to-Speech, Moderasi, dan kunci API terdaftar. Integrasi Cloudflare Quick Tunnel dan dukungan proxy cloud untuk akses jarak jauh.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -Membuat, mencakup, dan mencabut kunci API. Setiap kunci dapat dibatasi untuk model/penyedia tertentu dengan akses penuh atau izin hanya baca. Manajemen kunci visual dengan pelacakan penggunaan.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Pelacakan tindakan administratif dengan pemfilteran berdasarkan jenis tindakan, aktor, target, alamat IP, dan stempel waktu. Riwayat peristiwa keamanan penuh.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Aplikasi desktop Native Electron untuk Windows, macOS, dan Linux. Jalankan OmniRoute sebagai aplikasi mandiri dengan integrasi baki sistem, dukungan offline, pembaruan otomatis, dan instalasi sekali klik. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Fitur utama: +Key features: -- Polling kesiapan server (tidak ada layar kosong saat cold start) -- Baki sistem dengan manajemen port -- Kebijakan Keamanan Konten -- Kunci instans tunggal -- Pembaruan otomatis saat restart -- UI bersyarat platform (lampu lalu lintas macOS, bilah judul default Windows/Linux) -- Pengemasan build Hardened Electron — `node_modules` yang disinkronkan dalam bundel mandiri terdeteksi dan ditolak sebelum pengemasan, mencegah ketergantungan runtime pada mesin build (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Lihat [`electron/README.md`](../electron/README.md) untuk dokumentasi lengkap. +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/id/docs/TROUBLESHOOTING.md b/docs/i18n/id/docs/TROUBLESHOOTING.md index ea8f3ba0a0..e631f0b0c2 100644 --- a/docs/i18n/id/docs/TROUBLESHOOTING.md +++ b/docs/i18n/id/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -Masalah umum dan solusi untuk OmniRoute.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Masalah | Solusi | -| ----------------------------------------- | ----------------------------------------------------------------------- | --- | -| Login pertama tidak berfungsi | Setel `INITIAL_PASSWORD` di `.env` (tanpa hardcode default) | -| Dasbor terbuka pada port yang salah | Setel `PORT=20128` dan `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Tidak ada log permintaan di bawah `logs/` | Setel `ENABLE_REQUEST_LOGS=true` | -| EACCES: izin ditolak | Setel `DATA_DIR=/path/to/writable/dir` untuk mengganti `~/.omniroute` | -| Strategi perutean tidak menyimpan | Perbarui ke v1.4.11+ (perbaikan skema Zod untuk persistensi pengaturan) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Penyebab:**Kuota penyedia habis. +**Cause:** Provider quota exhausted. -**Perbaikan:** +**Fix:** -1. Periksa pelacak kuota dasbor -2. Gunakan kombo dengan tier fallback -3. Beralih ke tingkat yang lebih murah/gratis### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Penyebab:**Kuota berlangganan habis. +### Rate Limiting -**Perbaikan:** +**Cause:** Subscription quota exhausted. -- Tambahkan cadangan: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Gunakan GLM/MiniMax sebagai cadangan murah### OAuth Token Expired +**Fix:** -OmniRoute menyegarkan token secara otomatis. Jika masalah terus berlanjut: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Dasbor → Penyedia → Sambungkan kembali -2. Hapus dan tambahkan kembali koneksi penyedia--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Verifikasikan `BASE_URL` menunjuk ke instance Anda yang sedang berjalan (misalnya, `http://localhost:20128`) -2. Verifikasikan titik `CLOUD_URL` ke titik akhir cloud Anda (misalnya, `https://omniroute.dev`) -3. Jaga agar nilai `NEXT_PUBLIC_*` selaras dengan nilai sisi server### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Gejala:**`Token 'd'...` tak terduga di titik akhir cloud untuk panggilan non-streaming. +### Cloud `stream=false` Returns 500 -**Penyebab:**Upstream mengembalikan payload SSE sementara klien mengharapkan JSON. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Solusi:**Gunakan `stream=true` untuk panggilan cloud langsung. Waktu proses lokal mencakup penggantian SSE→JSON.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Buat kunci baru dari dasbor lokal (`/api/keys`) -2. Jalankan sinkronisasi cloud: Aktifkan Cloud → Sinkronkan Sekarang -3. Kunci lama/tidak disinkronkan masih dapat mengembalikan `401` di cloud--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Periksa kolom runtime: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. Untuk mode portabel: gunakan target gambar `runner-cli` (CLI yang dibundel) -3. Untuk mode pemasangan host: setel `CLI_EXTRA_PATHS` dan pasang direktori host bin sebagai hanya-baca -4. Jika `installed=true` dan `runnable=false`: biner ditemukan tetapi pemeriksaan kesehatan gagal### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Periksa statistik penggunaan di Dashboard → Penggunaan -2. Ganti model utama ke GLM/MiniMax -3. Gunakan tingkat gratis (Gemini CLI, Qoder) untuk tugas-tugas yang tidak penting -4. Tetapkan anggaran biaya per kunci API: Dasbor → Kunci API → Anggaran--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Setel `ENABLE_REQUEST_LOGS=true` di file `.env` Anda. Log muncul di bawah direktori `logs/`.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Status utama: `${DATA_DIR}/storage.sqlite` (penyedia, kombo, alias, kunci, pengaturan) -- Penggunaan: Tabel SQLite di `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + opsional `${DATA_DIR}/log.txt` dan `${DATA_DIR}/call_logs/` -- Log permintaan: `/logs/...` (bila `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Ketika pemutus arus penyedia TERBUKA, permintaan diblokir hingga cooldown berakhir. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Perbaikan:** +**Fix:** -1. Buka**Dasbor → Pengaturan → Ketahanan** -2. Periksa kartu pemutus arus untuk penyedia yang terpengaruh -3. Klik**Reset Semua**untuk menghapus semua pemutus, atau tunggu hingga cooldown berakhir -4. Pastikan penyedia benar-benar tersedia sebelum melakukan reset### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Jika penyedia berulang kali memasuki status OPEN: +### Provider keeps tripping the circuit breaker -1. Periksa**Dasbor → Kesehatan → Kesehatan Penyedia**untuk mengetahui pola kegagalannya -2. Buka**Pengaturan → Ketahanan → Profil Penyedia**dan tingkatkan ambang kegagalan -3. Periksa apakah penyedia telah mengubah batas API atau memerlukan autentikasi ulang -4. Tinjau telemetri latensi — latensi tinggi dapat menyebabkan kegagalan berbasis waktu habis--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Pastikan Anda menggunakan awalan yang benar: `deepgram/nova-3` atau `assemblyai/best` -- Verifikasi penyedia terhubung di**Dasbor → Penyedia**### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Periksa format audio yang didukung: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- Pastikan ukuran file berada dalam batas penyedia (biasanya <25MB) -- Periksa validitas kunci API penyedia di kartu penyedia--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Gunakan**Dasbor → Penerjemah**untuk men-debug masalah terjemahan format: +Use **Dashboard → Translator** to debug format translation issues: -| Modus | Kapan Menggunakan | -| -------------------- | -------------------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Taman Bermain** | Bandingkan format masukan/keluaran secara berdampingan — tempelkan permintaan yang gagal untuk melihat terjemahannya | -| **Penguji Obrolan** | Kirim pesan langsung dan periksa muatan permintaan/respons lengkap termasuk header | -| **Bangku Tes** | Jalankan pengujian batch di seluruh kombinasi format untuk menemukan terjemahan mana yang rusak | -| **Monitor Langsung** | Tonton alur permintaan waktu nyata untuk mengetahui masalah terjemahan yang terputus-putus | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Tag berpikir tidak muncul**— Periksa apakah penyedia target mendukung pemikiran dan pengaturan anggaran pemikiran -**Panggilan alat terputus**— Beberapa terjemahan format mungkin menghapus bidang yang tidak didukung; verifikasi dalam mode Taman Bermain -**Perintah sistem hilang**— Claude dan Gemini menangani perintah sistem secara berbeda; periksa keluaran terjemahan -**SDK mengembalikan string mentah, bukan objek**— Diperbaiki di v1.1.0: pembersih respons sekarang menghapus kolom non-standar (`x_groq`, `usage_breakdown`, dll.) yang menyebabkan kegagalan validasi OpenAI SDK Pydantic -**GLM/ERNIE menolak peran `sistem`**— Diperbaiki di v1.1.0: penormal peran secara otomatis menggabungkan pesan sistem ke dalam pesan pengguna untuk model yang tidak kompatibel -**`peran pengembang` tidak dikenali**— Diperbaiki di v1.1.0: otomatis dikonversi ke `sistem` untuk penyedia non-OpenAI -**`json_schema` tidak berfungsi dengan Gemini**— Diperbaiki di v1.1.0: `response_format` sekarang dikonversi ke `responseMimeType` + `responseSchema` Gemini--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- Batas tarif otomatis hanya berlaku untuk penyedia kunci API (bukan OAuth/langganan) -- Verifikasi**Pengaturan → Ketahanan → Profil Penyedia**telah mengaktifkan batas tarif otomatis -- Periksa apakah penyedia mengembalikan kode status `429` atau header `Retry-After`### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -Profil penyedia mendukung pengaturan berikut: +### Tuning exponential backoff --**Penundaan dasar**— Waktu tunggu awal setelah kegagalan pertama (default: 1 detik) -**Penundaan maksimal**— Batas waktu tunggu maksimum (default: 30 detik) -**Pengganda**— Berapa banyak peningkatan penundaan per kegagalan berturut-turut (default: 2x)### Anti-thundering herd +Provider profiles support these settings: -Ketika banyak permintaan bersamaan mencapai penyedia dengan tarif terbatas, OmniRoute menggunakan mutex + pembatasan tarif otomatis untuk membuat serialisasi permintaan dan mencegah kegagalan berjenjang. Ini otomatis untuk penyedia kunci API.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Beberapa pengguna OmniRoute menempatkan gateway di depan tumpukan RAG atau agen. Dalam pengaturan tersebut, biasanya terlihat pola yang aneh: OmniRoute terlihat sehat (penyedia aktif, profil perutean baik-baik saja, tidak ada peringatan batas tarif) tetapi jawaban akhirnya masih salah. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -Dalam praktiknya, insiden ini biasanya berasal dari pipeline RAG hilir, bukan dari gateway itu sendiri. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Jika Anda ingin kosakata bersama untuk menjelaskan kegagalan tersebut, Anda dapat menggunakan WFGY ProblemMap, sumber teks lisensi MIT eksternal yang mendefinisikan enam belas pola kegagalan RAG / LLM yang berulang. Pada tingkat tinggi mencakup: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- penyimpangan pengambilan dan batas konteks yang rusak -- indeks dan penyimpanan vektor kosong atau basi -- penyematan versus ketidakcocokan semantik -- masalah perakitan cepat dan jendela konteks -- logika runtuh dan jawaban terlalu percaya diri -- kegagalan koordinasi rantai panjang dan agen -- memori multi agen dan penyimpangan peran -- masalah penerapan dan pemesanan bootstrap +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -Idenya sederhana: +The idea is simple: -1. Saat Anda menyelidiki respons yang buruk, catatlah: - - tugas dan permintaan pengguna - - kombo rute atau penyedia di OmniRoute - - konteks RAG apa pun yang digunakan di hilir (dokumen yang diambil, panggilan alat, dll) -2. Petakan kejadian ke satu atau dua nomor Peta Masalah WFGY (`No.1` … `No.16`). -3. Simpan nomor tersebut di dasbor, runbook, atau pelacak insiden Anda sendiri di samping log OmniRoute. -4. Gunakan halaman WFGY yang sesuai untuk memutuskan apakah Anda perlu mengubah tumpukan RAG, retriever, atau strategi perutean Anda. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -Teks lengkap dan resep konkret ada di sini (lisensi MIT, hanya teks): +Full text and concrete recipes live here (MIT license, text only): -[README Peta Masalah WFGY](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) +[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -Anda dapat mengabaikan bagian ini jika Anda tidak menjalankan RAG atau alur agen di belakang OmniRoute.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**Masalah GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Arsitektur**: Lihat [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) untuk detail internal -**Referensi API**: Lihat [`docs/API_REFERENCE.md`](API_REFERENCE.md) untuk semua titik akhir -**Dasbor Kesehatan**: Periksa**Dasbor → Kesehatan**untuk status sistem waktu nyata -**Penerjemah**: Gunakan**Dasbor → Penerjemah**untuk men-debug masalah format +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/id/llm.txt b/docs/i18n/id/llm.txt new file mode 100644 index 0000000000..28f28cd8d5 --- /dev/null +++ b/docs/i18n/id/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Bahasa Indonesia) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Ikhtisar + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Keamanan +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/in/README.md b/docs/i18n/in/README.md index 5de9681b6e..d38dcdb52d 100644 --- a/docs/i18n/in/README.md +++ b/docs/i18n/in/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_आपका सार्वभौमिक एपीआई प्रॉक्सी - एक समापन बिंदु, 60+ प्रदाता, शून्य डाउनटाइम। अब**एमसीपी सर्वर (25 टूल्स)**,**ए2ए प्रोटोकॉल**,**मेमोरी/स्किल सिस्टम**और**इलेक्ट्रॉन डेस्कटॉप ऐप**के साथ।_ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**चैट समापन • एंबेडिंग • छवि निर्माण • वीडियो • संगीत • ऑडियो • पुनःरैंकिंग •**वेब खोज**• एमसीपी सर्वर • ए2ए प्रोटोकॉल • 100% टाइपस्क्रिप्ट**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _आपका सार्वभौमिक एपीआई प्रॉक् [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 वेबसाइट](https://omniroute.online) • [🚀 त्वरित प्रारंभ](#-त्वरित-प्रारंभ) • [💡 विशेषताएं](#-कुंजी-विशेषताएं) • [📖 दस्तावेज़](#-दस्तावेज़ीकरण) • [💰 मूल्य निर्धारण](#-मूल्य निर्धारण-एक नजर में) • [💬 व्हाट्सएप](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**इसमें उपलब्ध:**🇺🇸 [अंग्रेजी](README.md) | 🇧🇷 [पुर्तगाली (ब्राजील)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [फ़्रांसीसी](docs/i18n/fr/README.md) | 🇮🇹 [इतालवी](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [जर्मन](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [डांस्क](docs/i18n/da/README.md) | 🇫🇮 [सुओमी](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [मग्यार](docs/i18n/hu/README.md) | 🇮🇩 [बहासा इंडोनेशिया](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [बहासा मेलायु](docs/i18n/ms/README.md) | 🇳🇱 [नीदरलैंड्स](docs/i18n/nl/README.md) | 🇳🇴 [नॉर्स्क](docs/i18n/no/README.md) | 🇵🇹 [पुर्तगाली (पुर्तगाल)](docs/i18n/pt/README.md) | 🇷🇴 [रोमानिया](docs/i18n/ro/README.md) | 🇵🇱 [पोल्स्की](docs/i18n/pl/README.md) | 🇸🇰[स्लोवेनसीना](docs/i18n/sk/README.md) | 🇸🇪 [स्वेन्स्का](docs/i18n/sv/README.md) | 🇵🇭 [फ़िलिपिनो](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,553 +60,629 @@ _आपका सार्वभौमिक एपीआई प्रॉक् ## 📸 Dashboard Preview -<विवरण> -<सारांश>डैशबोर्ड स्क्रीनशॉट देखने के लिए क्लिक करें +
+Click to see dashboard screenshots -| पेज | स्क्रीनशॉट | -| ---------------- | ------------------------------------------------------------ | ---------- | -| **प्रदाता** | ![प्रदाता](docs/screenshots/01-providers.png) | -| **कॉम्बोस** | ![कॉम्बोस](दस्तावेज़/स्क्रीनशॉट/02-कॉम्बोस.पीएनजी) | -| **एनालिटिक्स** | ![एनालिटिक्स](docs/screenshots/03-analytics.png) | -| **स्वास्थ्य** | ![स्वास्थ्य](docs/screenshots/04-health.png) | -| **अनुवादक** | ![अनुवादक](docs/screenshots/05-translator.png) | -| **सेटिंग्स** | ![सेटिंग्स](दस्तावेज़/स्क्रीनशॉट/06-सेटिंग्स.पीएनजी) | -| **सीएलआई उपकरण** | ![सीएलआई उपकरण](दस्तावेज़/स्क्रीनशॉट/07-सीएलआई-टूल्स.पीएनजी) | -| **उपयोग लॉग** | ![उपयोग](docs/screenshots/08-usage.png) | -| **अंतबिंदु** | ![समाप्ति बिंदु](दस्तावेज़/स्क्रीनशॉट/09-endpoint.png) |
| +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | + + --- ### 🤖 Free AI Provider for your favorite coding agents -_OmniRoute के माध्यम से किसी भी AI-संचालित IDE या CLI टूल को कनेक्ट करें - असीमित कोडिंग के लिए निःशुल्क API गेटवे।_ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ -<तालिका> - - - -OpenClaw
-ओपनक्लॉ -

-⭐ 205K - - - -NanoBot
-नैनोबॉट -

-⭐ 20.9K - - - -”PicoClaw”
-पिकोक्लॉ -

-⭐ 14.6K - - - -ZeroClaw
-जीरोक्लॉ -

-⭐ 9.9K - - - -IronClaw
-आयरनक्लॉ -

-⭐ 2.1K - - - - - -OpenCode
-ओपनकोड -

-⭐ 106K - - - -कोडेक्स सीएलआई
-कोडेक्स सीएलआई -

-⭐ 60.8K - - - -क्लाउड कोड
-क्लाउड कोड -

-⭐ 67.3K - - - -मिथुन सीएलआई
-मिथुन सीएलआई -

-⭐ 94.7K - - - -किलो कोड
-किलो कोड -

-⭐ 15.5K - - - + + + + + + + + + + + + + + + +
+ + OpenClaw
+ OpenClaw +

+ ⭐ 205K +
+ + NanoBot
+ NanoBot +

+ ⭐ 20.9K +
+ + PicoClaw
+ PicoClaw +

+ ⭐ 14.6K +
+ + ZeroClaw
+ ZeroClaw +

+ ⭐ 9.9K +
+ + IronClaw
+ IronClaw +

+ ⭐ 2.1K +
+ + OpenCode
+ OpenCode +

+ ⭐ 106K +
+ + Codex CLI
+ Codex CLI +

+ ⭐ 60.8K +
+ + Claude Code
+ Claude Code +

+ ⭐ 67.3K +
+ + Gemini CLI
+ Gemini CLI +

+ ⭐ 94.7K +
+ + Kilo Code
+ Kilo Code +

+ ⭐ 15.5K +
-📡 सभी एजेंट http://localhost:20128/v1 या http://cloud.omniroute.online/v1 के माध्यम से जुड़ते हैं - एक कॉन्फ़िगरेशन, असीमित मॉडल और कोटा--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**पैसा बर्बाद करना और सीमा पार करना बंद करें:** +**Stop wasting money and hitting limits:** -- सदस्यता कोटा हर महीने अप्रयुक्त रूप से समाप्त हो जाता है -- दर सीमा आपको कोडिंग के बीच में रोक देती है -- महंगे एपीआई ($20-50/माह प्रति प्रदाता) -- प्रदाताओं के बीच मैन्युअल स्विचिंग +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute इसका समाधान करता है:** +**OmniRoute solves this:** -- ✅**सब्सक्रिप्शन अधिकतम करें**- कोटा ट्रैक करें, रीसेट से पहले हर बिट का उपयोग करें -- ✅**ऑटो फ़ॉलबैक**- सदस्यता → एपीआई कुंजी → सस्ता → निःशुल्क, शून्य डाउनटाइम -- ✅**मल्टी-अकाउंट**- प्रति प्रदाता खातों के बीच राउंड-रॉबिन -- ✅**यूनिवर्सल**- क्लाउड कोड, कोडेक्स, जेमिनी सीएलआई, कर्सर, क्लाइन, ओपनक्लॉ, किसी भी सीएलआई टूल के साथ काम करता है--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**हमारे समुदाय में शामिल हों!**[व्हाट्सएप ग्रुप](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) - सहायता प्राप्त करें, टिप्स साझा करें और अपडेट रहें। +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**वेबसाइट**: [omniroute.online](https://omniroute.online) -**गिटहब**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**मुद्दे**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**व्हाट्सएप**: [सामुदायिक समूह](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**योगदान**: [CONTRIBUTING.md](CONTRIBUTING.md) देखें, एक पीआर खोलें, या एक `अच्छा पहला अंक` चुनें -**मूल परियोजना**: [डेकोलुआ द्वारा 9राउटर](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -कोई समस्या खोलते समय, कृपया सिस्टम-जानकारी कमांड चलाएँ और जेनरेट की गई फ़ाइल संलग्न करें:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -यह आपके Node.js संस्करण, ओमनीरूट संस्करण, ओएस विवरण, स्थापित सीएलआई उपकरण (क्यूडर, जेमिनी, क्लाउड, कोडेक्स, एंटीग्रेविटी, ड्रॉइड, आदि), डॉकर/पीएम2 स्थिति और सिस्टम पैकेज के साथ एक `system-info.txt` उत्पन्न करता है - वह सब कुछ जो हमें आपकी समस्या को शीघ्रता से पुन: उत्पन्न करने के लिए चाहिए। फ़ाइल को सीधे अपने GitHub मुद्दे से संलग्न करें।--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**एआई टूल का उपयोग करने वाला प्रत्येक डेवलपर प्रतिदिन इन समस्याओं का सामना करता है।**ओम्नीरूट को उन सभी को हल करने के लिए बनाया गया था - लागत वृद्धि से लेकर क्षेत्रीय ब्लॉक तक, टूटे हुए ओएथ प्रवाह से लेकर प्रोटोकॉल संचालन और एंटरप्राइज़ अवलोकन तक। +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. -<विवरण> -<सारांश>💸 1. "मैं एक महंगी सदस्यता के लिए भुगतान करता हूं लेकिन फिर भी सीमा से बाधित होता हूं" +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -डेवलपर्स क्लाउड प्रो, कोडेक्स प्रो, या गिटहब कोपायलट के लिए $20-200/माह का भुगतान करते हैं। यहां तक ​​कि भुगतान करने पर भी, कोटा की एक सीमा होती है - 5 घंटे का उपयोग, साप्ताहिक सीमा, या प्रति मिनट की दर सीमा। मध्य-कोडिंग सत्र में, प्रदाता प्रत्युत्तर देना बंद कर देता है और डेवलपर प्रवाह और उत्पादकता खो देता है। +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**ओम्नीरूट इसे कैसे हल करता है:** +**How OmniRoute solves it:** --**स्मार्ट 4-टियर फ़ॉलबैक**- यदि सदस्यता कोटा समाप्त हो जाता है, तो स्वचालित रूप से एपीआई कुंजी पर रीडायरेक्ट हो जाता है → सस्ता → शून्य मैन्युअल हस्तक्षेप के साथ मुफ़्त --**प्रदाता ट्रैकिंग सीमाएं**- कैश्ड कोटा स्नैपशॉट सर्वर-साइड शेड्यूल पर रीफ्रेश होता है (डिफ़ॉल्ट `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) यूआई में मैन्युअल रीफ्रेश के साथ उपलब्ध है --**मल्टी-अकाउंट सपोर्ट**- ऑटो राउंड-रॉबिन के साथ प्रति प्रदाता एकाधिक खाते - जब एक खत्म हो जाता है, तो अगले पर स्विच हो जाता है --**कस्टम कॉम्बो**- 9 संतुलन रणनीतियों (प्राथमिकता, भारित, भरण-प्रथम, राउंड-रॉबिन, पी2सी, यादृच्छिक, कम से कम उपयोग, लागत-अनुकूलित, सख्त-यादृच्छिक) के साथ अनुकूलन योग्य फ़ॉलबैक चेन --**कोडेक्स बिजनेस कोटा**- बिजनेस/टीम कार्यक्षेत्र कोटा की निगरानी सीधे डैशबोर्ड में
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard -<विवरण> -<सारांश>🔌 2. "मुझे कई प्रदाताओं का उपयोग करने की आवश्यकता है लेकिन प्रत्येक के पास एक अलग एपीआई है" + -ओपनएआई एक प्रारूप का उपयोग करता है, क्लाउड (एंथ्रोपिक) दूसरे का उपयोग करता है, जेमिनी एक और का उपयोग करता है। यदि कोई डेवलपर विभिन्न प्रदाताओं के मॉडल का परीक्षण करना चाहता है या उनके बीच फ़ॉलबैक करना चाहता है, तो उन्हें एसडीके को फिर से कॉन्फ़िगर करना होगा, एंडपॉइंट बदलना होगा, असंगत प्रारूपों से निपटना होगा। कस्टम प्रदाताओं (फ्रेंडएलआई, एनआईएम) के पास गैर-मानक मॉडल एंडपॉइंट हैं। +
+🔌 2. "I need to use multiple providers but each has a different API" -**ओम्नीरूट इसे कैसे हल करता है:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**एकीकृत समापन बिंदु**- एक एकल `http://localhost:20128/v1` सभी 60+ प्रदाताओं के लिए प्रॉक्सी के रूप में कार्य करता है --**प्रारूप अनुवाद**- स्वचालित और पारदर्शी: ओपनएआई ↔ क्लाउड ↔ जेमिनी ↔ प्रतिक्रिया एपीआई --**प्रतिक्रिया स्वच्छता**- गैर-मानक फ़ील्ड (`x_groq`, `usage_breakdown`, `service_tier`) को हटा दें जो OpenAI SDK v1.83+ को तोड़ता है --**भूमिका सामान्यीकरण**- गैर-ओपनएआई प्रदाताओं के लिए `डेवलपर` → `सिस्टम` को रूपांतरित करता है; GLM/ERNIE के लिए `सिस्टम` → `उपयोगकर्ता` --**थिंक टैग एक्सट्रैक्शन**- डीपसीक आर1 जैसे मॉडलों से `<थिंक>` ब्लॉक को मानकीकृत `रीज़निंग_कंटेंट` में निकाला जाता है --**मिथुन के लिए संरचित आउटपुट**- `json_schema` → `responseMimeType`/`responseSchema` स्वचालित रूपांतरण --**`स्ट्रीम` डिफ़ॉल्ट रूप से `झूठा`**होता है - ओपनएआई स्पेक के साथ संरेखित होता है, पायथन/रस्ट/गो एसडीके में अप्रत्याशित एसएसई से बचता है
+**How OmniRoute solves it:** -<विवरण> -<सारांश>🌐 3. "मेरा AI प्रदाता मेरे क्षेत्र/देश को ब्लॉक कर देता है" +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -OpenAI/Codex जैसे प्रदाता कुछ भौगोलिक क्षेत्रों से पहुंच को रोकते हैं। उपयोगकर्ताओं को OAuth और API कनेक्शन के दौरान `unsupported_country_region_territory` जैसी त्रुटियां मिलती हैं। यह विकासशील देशों के डेवलपर्स के लिए विशेष रूप से निराशाजनक है। + -**ओम्नीरूट इसे कैसे हल करता है:** +
+🌐 3. "My AI provider blocks my region/country" --**3-स्तरीय प्रॉक्सी कॉन्फ़िगरेशन**- 3 स्तरों पर कॉन्फ़िगर करने योग्य प्रॉक्सी: वैश्विक (सभी ट्रैफ़िक), प्रति-प्रदाता (केवल एक प्रदाता), और प्रति-कनेक्शन/कुंजी --**रंग-कोडित प्रॉक्सी बैज**- दृश्य संकेतक: 🟢 वैश्विक प्रॉक्सी, 🟡 प्रदाता प्रॉक्सी, 🔵 कनेक्शन प्रॉक्सी, हमेशा आईपी दिखाता है --**प्रॉक्सी के माध्यम से OAuth टोकन एक्सचेंज**- OAuth प्रवाह भी प्रॉक्सी के माध्यम से जाता है, `unsupported_country_region_territory` को हल करता है --**प्रॉक्सी के माध्यम से कनेक्शन परीक्षण**- कनेक्शन परीक्षण कॉन्फ़िगर प्रॉक्सी का उपयोग करते हैं (अब कोई प्रत्यक्ष बाईपास नहीं) --**SOCKS5 समर्थन**- आउटबाउंड रूटिंग के लिए पूर्ण SOCKS5 प्रॉक्सी समर्थन --**टीएलएस फ़िंगरप्रिंट स्पूफिंग**- बॉट डिटेक्शन को बायपास करने के लिए `wreq-js` के माध्यम से ब्राउज़र जैसा टीएलएस फ़िंगरप्रिंट --**🔏 सीएलआई फ़िंगरप्रिंट मिलान**- मूल सीएलआई बाइनरी हस्ताक्षरों से मिलान करने के लिए हेडर और बॉडी फ़ील्ड को पुन: व्यवस्थित करता है, जिससे खाता फ़्लैगिंग जोखिम काफी कम हो जाता है। प्रॉक्सी आईपी संरक्षित है - आपको एक साथ स्टील्थ**और**आईपी मास्किंग दोनों मिलते हैं
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. -<विवरण> -<सारांश>🆓 4. "मैं कोडिंग के लिए AI का उपयोग करना चाहता हूं लेकिन मेरे पास पैसे नहीं हैं" +**How OmniRoute solves it:** -हर कोई AI सदस्यता के लिए $20-200/माह का भुगतान नहीं कर सकता। छात्रों, उभरते देशों के डेवलपर्स, शौकीनों और फ्रीलांसरों को शून्य लागत पर गुणवत्ता वाले मॉडल तक पहुंच की आवश्यकता है। +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**ओम्नीरूट इसे कैसे हल करता है:** + --**फ्री टियर प्रोवाइडर्स बिल्ट-इन**- 100% मुफ्त प्रदाताओं के लिए मूल समर्थन: क्यूडर (OAuth के माध्यम से 5 असीमित मॉडल: किमी-के2-थिंकिंग, क्यूवेन3-कोडर-प्लस, डीपसीक-आर1, मिनिमैक्स-एम2, किमी-के2), क्यूवेन (4 असीमित मॉडल: क्यूवेन3-कोडर-प्लस, क्यूवेन3-कोडर-फ्लैश, क्यूवेन3-कोडर-नेक्स्ट, विज़न-मॉडल), किरो (क्लाउड + एडब्ल्यूएस बिल्डर आईडी मुफ़्त), जेमिनी सीएलआई (180K टोकन/माह मुफ़्त) --**ओलामा क्लाउड**- निःशुल्क "लाइट उपयोग" स्तर के साथ `api.ollama.com` पर क्लाउड-होस्टेड ओलामा मॉडल; `ollamacloud/` उपसर्ग का उपयोग करें --**केवल-नि:शुल्क कॉम्बो**- चेन `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/माह शून्य डाउनटाइम के साथ --**एनवीडिया एनआईएम फ्री एक्सेस**- ~40 आरपीएम डेव-बिल्ड.एनवीडिया.कॉम पर 70+ मॉडलों तक हमेशा के लिए मुफ्त एक्सेस (क्रेडिट से शुद्ध दर सीमा तक संक्रमण) --**लागत अनुकूलित रणनीति**- रूटिंग रणनीति जो स्वचालित रूप से सबसे सस्ते उपलब्ध प्रदाता को चुनती है +
+🆓 4. "I want to use AI for coding but I have no money" -<विवरण> -<सारांश>🔒 5. "मुझे अपने AI गेटवे को अनधिकृत पहुंच से बचाने की आवश्यकता है" +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -नेटवर्क (LAN, VPS, Docker) में AI गेटवे को उजागर करते समय, पते वाला कोई भी व्यक्ति डेवलपर के टोकन/कोटा का उपभोग कर सकता है। सुरक्षा के बिना, एपीआई दुरुपयोग, त्वरित इंजेक्शन और दुरुपयोग के प्रति संवेदनशील हैं। +**How OmniRoute solves it:** -**ओम्नीरूट इसे कैसे हल करता है:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**एपीआई कुंजी प्रबंधन**- एक समर्पित `/डैशबोर्ड/एपीआई-मैनेजर` पेज के साथ प्रति प्रदाता जेनरेशन, रोटेशन और स्कोपिंग --**मॉडल-स्तरीय अनुमतियाँ**- सभी को अनुमति दें/प्रतिबंधित टॉगल के साथ एपीआई कुंजियों को विशिष्ट मॉडल (`ओपनाई/*`, वाइल्डकार्ड पैटर्न) तक सीमित करें --**एपीआई एंडपॉइंट सुरक्षा**- `/v1/मॉडल` के लिए एक कुंजी की आवश्यकता है और लिस्टिंग से विशिष्ट प्रदाताओं को ब्लॉक करें --**ऑथ गार्ड + सीएसआरएफ सुरक्षा**- सभी डैशबोर्ड रूट `withAuth` मिडलवेयर + सीएसआरएफ टोकन से सुरक्षित हैं --**रेट लिमिटर**- कॉन्फ़िगर करने योग्य विंडो के साथ प्रति-आईपी दर सीमित करना --**आईपी फ़िल्टरिंग**- अभिगम नियंत्रण के लिए अनुमति सूची/अवरुद्ध सूची --**प्रॉम्प्ट इंजेक्शन गार्ड**- दुर्भावनापूर्ण प्रॉम्प्ट पैटर्न के विरुद्ध स्वच्छता --**एईएस-256-जीसीएम एन्क्रिप्शन**- क्रेडेंशियल आराम से एन्क्रिप्ट किए गए
+ -<विवरण> -<सारांश>🛑 6. "मेरा प्रदाता बंद हो गया और मैंने अपना कोडिंग प्रवाह खो दिया" +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -एआई प्रदाता अस्थिर हो सकते हैं, 5xx त्रुटियाँ लौटा सकते हैं, या अस्थायी दर सीमा तक पहुँच सकते हैं। यदि कोई डेवलपर किसी एकल प्रदाता पर निर्भर करता है, तो वे बाधित हो जाते हैं। सर्किट ब्रेकर के बिना, बार-बार पुनः प्रयास करने से एप्लिकेशन क्रैश हो सकता है। +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**ओम्नीरूट इसे कैसे हल करता है:** +**How OmniRoute solves it:** --**प्रति-मॉडल सर्किट ब्रेकर**- कॉन्फ़िगर करने योग्य थ्रेसहोल्ड और कूलडाउन (बंद/खुला/आधा-खुला) के साथ ऑटो-खुला/बंद, कैस्केडिंग ब्लॉक से बचने के लिए प्रति-मॉडल स्कोप्ड --**एक्सपोनेंशियल बैकऑफ़**- प्रगतिशील पुनः प्रयास में देरी --**एंटी-थंडरिंग हर्ड**- म्यूटेक्स + समवर्ती रिट्री तूफानों के खिलाफ सेमाफोर सुरक्षा --**कॉम्बो फ़ॉलबैक चेन**- यदि प्राथमिक प्रदाता विफल हो जाता है, तो बिना किसी हस्तक्षेप के स्वचालित रूप से चेन से गिर जाता है --**कॉम्बो सर्किट ब्रेकर**- कॉम्बो श्रृंखला के भीतर विफल प्रदाताओं को स्वचालित रूप से अक्षम करता है --**स्वास्थ्य डैशबोर्ड**- अपटाइम मॉनिटरिंग, सर्किट ब्रेकर स्थिति, लॉकआउट, कैश आँकड़े, p50/p95/p99 विलंबता
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest -<विवरण> -<सारांश>🔧 7. "प्रत्येक AI उपकरण को कॉन्फ़िगर करना कठिन और दोहराव वाला है" + -डेवलपर्स कर्सर, क्लाउड कोड, कोडेक्स सीएलआई, ओपनक्लाव, जेमिनी सीएलआई, किलो कोड का उपयोग करते हैं... प्रत्येक टूल को एक अलग कॉन्फ़िगरेशन (एपीआई एंडपॉइंट, कुंजी, मॉडल) की आवश्यकता होती है। प्रदाताओं या मॉडलों को स्विच करते समय पुन: कॉन्फ़िगर करना समय की बर्बादी है। +
+🛑 6. "My provider went down and I lost my coding flow" -**ओम्नीरूट इसे कैसे हल करता है:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**सीएलआई टूल्स डैशबोर्ड**- क्लाउड कोड, कोडेक्स सीएलआई, ओपनक्लाव, किलो कोड, एंटीग्रेविटी, क्लाइन के लिए एक-क्लिक सेटअप वाला समर्पित पेज --**गिटहब कोपायलट कॉन्फिग जेनरेटर**- बल्क मॉडल चयन के साथ वीएस कोड के लिए `चैटलैंग्वेजमॉडल.जेसन` जेनरेट करता है --**ऑनबोर्डिंग विज़ार्ड**- पहली बार उपयोगकर्ताओं के लिए निर्देशित 4-चरणीय सेटअप --**एक समापन बिंदु, सभी मॉडल**- `http://localhost:20128/v1` को एक बार कॉन्फ़िगर करें, 60+ प्रदाताओं तक पहुंचें
+**How OmniRoute solves it:** -<विवरण> -<सारांश>🔑 8. "एकाधिक प्रदाताओं से OAuth टोकन प्रबंधित करना नरक है" +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -क्लाउड कोड, कोडेक्स, जेमिनी सीएलआई, कोपायलट - सभी समाप्त होने वाले टोकन के साथ OAuth 2.0 का उपयोग करते हैं। डेवलपर्स को लगातार पुन: प्रमाणित करने, `client_secret is missing`, `redirect_uri_mismatch` और दूरस्थ सर्वर पर विफलताओं से निपटने की आवश्यकता होती है। LAN/VPS पर OAuth विशेष रूप से समस्याग्रस्त है। + -**ओम्नीरूट इसे कैसे हल करता है:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**ऑटो टोकन रिफ्रेश**- OAuth टोकन समाप्ति से पहले पृष्ठभूमि में रिफ्रेश होते हैं --**OAuth 2.0 (PKCE) बिल्ट-इन**- क्लाउड कोड, कोडेक्स, जेमिनी सीएलआई, कोपायलट, किरो, क्वेन, कोडर के लिए स्वचालित प्रवाह --**मल्टी-अकाउंट OAuth**- JWT/ID टोकन निष्कर्षण के माध्यम से प्रति प्रदाता एकाधिक खाते --**OAuth LAN/रिमोट फिक्स**- `redirect_uri` के लिए निजी आईपी पहचान + रिमोट सर्वर के लिए मैनुअल यूआरएल मोड --**Nginx के पीछे OAuth**- रिवर्स प्रॉक्सी संगतता के लिए `window.location.origin` का उपयोग करता है --**दूरस्थ OAuth मार्गदर्शिका**— VPS/Docker पर Google क्लाउड क्रेडेंशियल के लिए चरण-दर-चरण मार्गदर्शिका
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. -<विवरण> -<सारांश>📊 9. "मुझे नहीं पता कि मैं कितना और कहां खर्च कर रहा हूं" +**How OmniRoute solves it:** -डेवलपर्स कई भुगतान प्रदाताओं का उपयोग करते हैं लेकिन खर्च के बारे में कोई एकीकृत दृष्टिकोण नहीं रखते हैं। प्रत्येक प्रदाता का अपना बिलिंग डैशबोर्ड होता है, लेकिन कोई समेकित दृश्य नहीं होता है। अप्रत्याशित लागतें बढ़ सकती हैं। +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**ओम्नीरूट इसे कैसे हल करता है:** + --**लागत विश्लेषण डैशबोर्ड**— प्रति प्रदाता प्रति टोकन लागत ट्रैकिंग और बजट प्रबंधन --**प्रति स्तर बजट सीमा**- प्रति स्तर खर्च की अधिकतम सीमा जो स्वचालित फ़ॉलबैक को ट्रिगर करती है --**प्रति-मॉडल मूल्य निर्धारण कॉन्फ़िगरेशन**- प्रति मॉडल कॉन्फ़िगर करने योग्य कीमतें --**प्रति एपीआई कुंजी उपयोग सांख्यिकी**- अनुरोध गणना और प्रति कुंजी अंतिम बार उपयोग किया गया टाइमस्टैम्प --**एनालिटिक्स डैशबोर्ड**- स्टेट कार्ड, मॉडल उपयोग चार्ट, सफलता दर और विलंबता के साथ प्रदाता तालिका +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" -<विवरण> -<सारांश>🐛 10. "मैं एआई कॉल में त्रुटियों और समस्याओं का निदान नहीं कर सकता" +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -जब कोई कॉल विफल हो जाती है, तो देव को पता नहीं चलता कि यह दर सीमा, समाप्त टोकन, गलत प्रारूप या प्रदाता त्रुटि थी। विभिन्न टर्मिनलों पर खंडित लॉग। अवलोकन के बिना, डिबगिंग परीक्षण-और-त्रुटि है। +**How OmniRoute solves it:** -**ओम्नीरूट इसे कैसे हल करता है:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**एकीकृत लॉग डैशबोर्ड**- 4 टैब: अनुरोध लॉग, प्रॉक्सी लॉग, ऑडिट लॉग, कंसोल --**कंसोल लॉग व्यूअर**- रंग-कोडित स्तरों, ऑटो-स्क्रॉल, खोज, फ़िल्टर के साथ वास्तविक समय टर्मिनल-शैली व्यूअर --**SQLite प्रॉक्सी लॉग्स**- लगातार लॉग जो सर्वर पुनरारंभ होने से बचे रहते हैं --**अनुवादक खेल का मैदान**- 4 डिबगिंग मोड: खेल का मैदान (प्रारूप अनुवाद), चैट टेस्टर (राउंड-ट्रिप), टेस्ट बेंच (बैच), लाइव मॉनिटर (वास्तविक समय) --**अनुरोध टेलीमेट्री**- p50/p95/p99 विलंबता + X-अनुरोध-आईडी ट्रेसिंग --**रोटेशन के साथ फ़ाइल-आधारित लॉगिंग**- ऐप लॉग आकार, अवधारण दिनों और संग्रह गणना के अनुसार घूमते हैं; कॉल लॉग कलाकृतियाँ अवधारण दिनों और फ़ाइल गणना के अनुसार घूमती हैं --**सिस्टम जानकारी रिपोर्ट**- `npm run system-info` आपके पूर्ण वातावरण (नोड संस्करण, ओमनीरूट संस्करण, ओएस, सीएलआई उपकरण, डॉकर/पीएम2 स्थिति) के साथ `system-info.txt` उत्पन्न करता है। त्वरित ट्राइएज के लिए समस्याओं की रिपोर्ट करते समय इसे संलग्न करें।
+ -<विवरण> -<सारांश>🏗️ 11. "प्रवेश द्वार की तैनाती और रखरखाव जटिल है" +
+📊 9. "I don't know how much I'm spending or where" -विभिन्न वातावरणों (स्थानीय, वीपीएस, डॉकर, क्लाउड) में एआई प्रॉक्सी को स्थापित करना, कॉन्फ़िगर करना और बनाए रखना श्रम-गहन है। हार्डकोडेड पथ, निर्देशिकाओं पर `EACCES`, पोर्ट विरोध और क्रॉस-प्लेटफ़ॉर्म बिल्ड जैसी समस्याएं घर्षण बढ़ाती हैं। +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**ओम्नीरूट इसे कैसे हल करता है:** +**How OmniRoute solves it:** --**एनपीएम ग्लोबल इंस्टाल**- `एनपीएम इंस्टाल -जी ऑम्निरूटे && ऑम्निरूटे` - हो गया --**डॉकर मल्टी-प्लेटफ़ॉर्म**- AMD64 + ARM64 नेटिव (Apple सिलिकॉन, AWS ग्रेविटॉन, रास्पबेरी पाई) --**डॉकर कंपोज प्रोफाइल**- `बेस` (कोई सीएलआई उपकरण नहीं) और `सीएलआई` (क्लाउड कोड, कोडेक्स, ओपनक्लाव के साथ) --**इलेक्ट्रॉन डेस्कटॉप ऐप**- सिस्टम ट्रे, ऑटो-स्टार्ट, ऑफ़लाइन मोड के साथ विंडोज/मैकओएस/लिनक्स के लिए मूल ऐप --**स्प्लिट-पोर्ट मोड**- उन्नत परिदृश्यों के लिए अलग-अलग पोर्ट पर एपीआई और डैशबोर्ड (रिवर्स प्रॉक्सी, कंटेनर नेटवर्किंग) --**क्लाउड सिंक**- क्लाउडफ्लेयर वर्कर्स के माध्यम से सभी डिवाइसों में कॉन्फिग सिंक्रोनाइजेशन --**डीबी बैकअप**- बाह्य रूप से प्रबंधित बैकअप के लिए `DISABLE_SQLITE_AUTO_BACKUP` के साथ सभी सेटिंग्स का स्वचालित बैकअप, पुनर्स्थापना, निर्यात और आयात
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency -<विवरण> -<सारांश>🌍 12. "इंटरफ़ेस केवल अंग्रेजी है और मेरी टीम अंग्रेजी नहीं बोलती है" + -गैर-अंग्रेजी भाषी देशों, विशेष रूप से लैटिन अमेरिका, एशिया और यूरोप में टीमें, केवल अंग्रेजी इंटरफेस के साथ संघर्ष करती हैं। भाषा बाधाएँ अपनाने को कम करती हैं और कॉन्फ़िगरेशन त्रुटियों को बढ़ाती हैं। +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**ओम्नीरूट इसे कैसे हल करता है:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**डैशबोर्ड i18n - 30 भाषाएँ**- अरबी, बल्गेरियाई, डेनिश, जर्मन, स्पेनिश, फिनिश, फ्रेंच, हिब्रू, हिंदी, हंगेरियन, इंडोनेशियाई, इतालवी, जापानी, कोरियाई, मलय, डच, नॉर्वेजियन, पोलिश, पुर्तगाली (पीटी/बीआर), रोमानियाई, रूसी, स्लोवाक, स्वीडिश, थाई, यूक्रेनी, वियतनामी, चीनी, फिलिपिनो, अंग्रेजी सहित सभी 500+ कुंजियाँ अनुवादित --**आरटीएल समर्थन**- अरबी और हिब्रू के लिए दाएं से बाएं समर्थन --**बहु-भाषा रीडमी**- 30 पूर्ण दस्तावेज़ीकरण अनुवाद --**भाषा चयनकर्ता**- वास्तविक समय स्विचिंग के लिए हेडर में ग्लोब आइकन
+**How OmniRoute solves it:** -<विवरण> -<सारांश>🔄 13. "मुझे चैट से अधिक की आवश्यकता है - मुझे एम्बेडिंग, चित्र, ऑडियो की आवश्यकता है" +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -एआई का मतलब सिर्फ चैट पूरा करना नहीं है। डेवलपर्स को छवियां उत्पन्न करने, ऑडियो ट्रांसक्राइब करने, आरएजी के लिए एम्बेडिंग बनाने, दस्तावेज़ों को फिर से रैंक करने और सामग्री को मॉडरेट करने की आवश्यकता होती है। प्रत्येक एपीआई का एक अलग समापन बिंदु और प्रारूप होता है। + -**ओम्नीरूट इसे कैसे हल करता है:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**एंबेडिंग**- `/v1/एंबेडिंग` 6 प्रदाताओं और 9+ मॉडल के साथ --**छवि निर्माण**- 10 प्रदाताओं और 20+ मॉडलों के साथ `/v1/छवियां/पीढ़ी` (ओपनएआई, एक्सएआई, टुगेदर, फायरवर्क्स, नेबियस, हाइपरबोलिक, नैनोबनाना, एंटीग्रेविटी, एसडी वेबयूआई, कॉम्फीयूआई) --**टेक्स्ट-टू-वीडियो**- `/v1/वीडियो/पीढ़ी` - कॉम्फीयूआई (एनिमेटडिफ, एसवीडी) और एसडी वेबयूआई --**टेक्स्ट-टू-म्यूजिक**- `/v1/म्यूजिक/पीढ़ी` - कॉम्फीयूआई (स्थिर ऑडियो ओपन, म्यूजिकजेन) --**ऑडियो ट्रांसक्रिप्शन**- `/v1/ऑडियो/ट्रांसक्रिप्शन` - व्हिस्पर + एनवीडिया एनआईएम, हगिंगफेस, क्वेन3 --**टेक्स्ट-टू-स्पीच**- `/v1/ऑडियो/स्पीच` - इलेवनलैब्स, एनवीडिया एनआईएम, हगिंगफेस, कोक्वी, टोरटोइज़, क्वेन3,**इनवर्ल्ड**,**कार्टेसिया**,**प्लेएचटी**, + मौजूदा प्रदाता --**मॉडरेशन**- `/v1/मॉडरेशन` - सामग्री सुरक्षा जांच --**रीरैंकिंग**— `/v1/rerank` — दस्तावेज़ प्रासंगिकता रीरैंकिंग --**प्रतिक्रिया एपीआई**- कोडेक्स के लिए पूर्ण `/v1/प्रतिक्रिया` समर्थन
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. -<विवरण> -<सारांश>🧪 14. "मेरे पास सभी मॉडलों की गुणवत्ता का परीक्षण और तुलना करने का कोई तरीका नहीं है" +**How OmniRoute solves it:** -डेवलपर्स जानना चाहते हैं कि उनके उपयोग के मामले में कौन सा मॉडल सबसे अच्छा है - कोड, अनुवाद, तर्क - लेकिन मैन्युअल रूप से तुलना करना धीमा है। कोई एकीकृत eval उपकरण मौजूद नहीं है। +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**ओम्नीरूट इसे कैसे हल करता है:** + --**एलएलएम मूल्यांकन**- अभिवादन, गणित, भूगोल, कोड जनरेशन, JSON अनुपालन, अनुवाद, मार्कडाउन, सुरक्षा इनकार को कवर करने वाले 10 प्री-लोडेड मामलों के साथ गोल्डन सेट परीक्षण --**4 मिलान रणनीतियाँ**- `सटीक`, `शामिल`, `रेगेक्स`, `कस्टम` (जेएस फ़ंक्शन) --**अनुवादक खेल का मैदान परीक्षण बेंच**- एकाधिक इनपुट और अपेक्षित आउटपुट, क्रॉस-प्रदाता तुलना के साथ बैच परीक्षण --**चैट परीक्षक**- दृश्य प्रतिक्रिया प्रतिपादन के साथ पूर्ण राउंड-ट्रिप --**लाइव मॉनिटर**- प्रॉक्सी के माध्यम से बहने वाले सभी अनुरोधों की वास्तविक समय स्ट्रीम +
+🌍 12. "The interface is English-only and my team doesn't speak English" -<विवरण> -<सारांश>📈 15. "मुझे प्रदर्शन खोए बिना स्केल करने की आवश्यकता है" +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -जैसे-जैसे अनुरोध की मात्रा बढ़ती है, कैशिंग के बिना वही प्रश्न डुप्लिकेट लागत उत्पन्न करते हैं। निष्क्रियता के बिना, डुप्लिकेट अपशिष्ट प्रसंस्करण का अनुरोध करता है। प्रति-प्रदाता दर सीमा का सम्मान किया जाना चाहिए। +**How OmniRoute solves it:** -**ओम्नीरूट इसे कैसे हल करता है:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**सिमेंटिक कैश**- दो-स्तरीय कैश (हस्ताक्षर + सिमेंटिक) लागत और विलंबता को कम करता है --**अनुरोध Idempotency**- समान अनुरोधों के लिए 5s डिडुप्लीकेशन विंडो --**दर सीमा का पता लगाना**- प्रति-प्रदाता आरपीएम, न्यूनतम अंतर, और अधिकतम समवर्ती ट्रैकिंग --**संपादन योग्य दर सीमाएँ**— सेटिंग्स में कॉन्फ़िगर करने योग्य डिफ़ॉल्ट → दृढ़ता के साथ लचीलापन --**एपीआई कुंजी सत्यापन कैश**- उत्पादन प्रदर्शन के लिए 3-स्तरीय कैश --**टेलीमेट्री के साथ स्वास्थ्य डैशबोर्ड**— p50/p95/p99 विलंबता, कैश आँकड़े, अपटाइम
+ -<विवरण> -<सारांश>🤖 16. "मैं विश्व स्तर पर मॉडल व्यवहार को नियंत्रित करना चाहता हूं" +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -ऐसे डेवलपर जो सभी प्रतिक्रियाएं एक विशिष्ट भाषा में, एक विशिष्ट लहजे में चाहते हैं, या तर्क टोकन को सीमित करना चाहते हैं। प्रत्येक टूल/अनुरोध में इसे कॉन्फ़िगर करना अव्यावहारिक है। +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**ओम्नीरूट इसे कैसे हल करता है:** +**How OmniRoute solves it:** --**सिस्टम प्रॉम्प्ट इंजेक्शन**— ग्लोबल प्रॉम्प्ट सभी अनुरोधों पर लागू होता है --**सोच बजट सत्यापन**- प्रति अनुरोध तर्क टोकन आवंटन नियंत्रण (पासथ्रू, ऑटो, कस्टम, अनुकूली) --**9 रूटिंग रणनीतियाँ**- वैश्विक रणनीतियाँ जो यह निर्धारित करती हैं कि अनुरोध कैसे वितरित किए जाते हैं --**वाइल्डकार्ड राउटर**- `प्रदाता/*` पैटर्न किसी भी प्रदाता को गतिशील रूप से रूट करता है --**कॉम्बो सक्षम/अक्षम टॉगल**— कॉम्बो को सीधे डैशबोर्ड से टॉगल करें --**प्रदाता टॉगल**— एक क्लिक से प्रदाता के लिए सभी कनेक्शन सक्षम/अक्षम करें --**अवरुद्ध प्रदाता**- `/v1/मॉडल` सूची से विशिष्ट प्रदाताओं को बाहर निकालें
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex -<विवरण> -<सारांश>🧰 17. "मुझे प्रथम श्रेणी उत्पाद क्षमताओं के रूप में एमसीपी टूल्स की आवश्यकता है" + -कई एआई गेटवे एमसीपी को केवल एक छिपे हुए कार्यान्वयन विवरण के रूप में उजागर करते हैं। टीमों को एक दृश्यमान, प्रबंधनीय संचालन परत की आवश्यकता होती है। +
+🧪 14. "I have no way to test and compare quality across models" -**ओम्नीरूट इसे कैसे हल करता है:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- एमसीपी डैशबोर्ड नेविगेशन और एंडपॉइंट प्रोटोकॉल टैब में दिखाई देता है -- प्रक्रिया, उपकरण, कार्यक्षेत्र और ऑडिट के साथ समर्पित एमसीपी प्रबंधन पृष्ठ -- `omniroute --mcp` और क्लाइंट ऑनबोर्डिंग के लिए बिल्ट-इन क्विक-स्टार्ट
+**How OmniRoute solves it:** -<विवरण> -<सारांश>🧠 18. "मुझे सिंक + स्ट्रीम कार्य पथों के साथ A2A ऑर्केस्ट्रेशन की आवश्यकता है" +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -एजेंट वर्कफ़्लो को जीवनचक्र नियंत्रण के साथ सीधे उत्तर और लंबे समय तक चलने वाले स्ट्रीम निष्पादन दोनों की आवश्यकता होती है। + -**ओम्नीरूट इसे कैसे हल करता है:** +
+📈 15. "I need to scale without losing performance" -- A2A JSON-RPC एंडपॉइंट (`POST /a2a`) `मैसेज/सेंड` और `मैसेज/स्ट्रीम` के साथ -- टर्मिनल राज्य प्रसार के साथ एसएसई स्ट्रीमिंग -- `कार्य/प्राप्त करें` और `कार्य/रद्द करें` के लिए कार्य जीवनचक्र एपीआई
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. -<विवरण> -<सारांश>🛰️ 19. "मुझे वास्तविक एमसीपी प्रक्रिया स्वास्थ्य की आवश्यकता है, अनुमानित स्थिति की नहीं" +**How OmniRoute solves it:** -परिचालन टीमों को यह जानने की जरूरत है कि क्या एमसीपी वास्तव में जीवित है, न कि केवल एपीआई पहुंच योग्य है या नहीं। +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**ओम्नीरूट इसे कैसे हल करता है:** + -- पीआईडी, टाइमस्टैम्प, ट्रांसपोर्ट, टूल काउंट और स्कोप मोड के साथ रनटाइम हार्टबीट फ़ाइल -- एमसीपी स्थिति एपीआई दिल की धड़कन + हाल की गतिविधि का संयोजन -- प्रक्रिया/अपटाइम/दिल की धड़कन ताजगी के लिए यूआई स्टेटस कार्ड +
+🤖 16. "I want to control model behavior globally" -<विवरण> -<सारांश>📋 20. "मुझे ऑडिटेबल एमसीपी टूल निष्पादन की आवश्यकता है" +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -जब उपकरण कॉन्फ़िगरेशन को बदलते हैं या ऑप्स क्रियाओं को ट्रिगर करते हैं, तो टीमों को फोरेंसिक ट्रैसेबिलिटी की आवश्यकता होती है। +**How OmniRoute solves it:** -**ओम्नीरूट इसे कैसे हल करता है:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- MCP टूल कॉल के लिए SQLite समर्थित ऑडिट लॉगिंग -- टूल, सफलता/असफलता, एपीआई कुंजी और पेजिनेशन द्वारा फ़िल्टर -- डैशबोर्ड ऑडिट टेबल + स्वचालन के लिए आँकड़े समापन बिंदु
+ -<विवरण> -<सारांश>🔐 21. "मुझे प्रति एकीकरण के लिए स्कोप्ड एमसीपी अनुमतियों की आवश्यकता है" +
+🧰 17. "I need MCP tools as first-class product capabilities" -विभिन्न ग्राहकों को टूल श्रेणियों तक कम से कम विशेषाधिकार प्राप्त होना चाहिए। +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**ओम्नीरूट इसे कैसे हल करता है:** +**How OmniRoute solves it:** -- नियंत्रित टूल एक्सेस के लिए 10 दानेदार एमसीपी स्कोप -- एमसीपी प्रबंधन यूआई में दायरा प्रवर्तन और दृश्यता -- परिचालन टूलींग के लिए सुरक्षित डिफ़ॉल्ट मुद्रा
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding -<विवरण> -<सारांश>⚙️ 22. "मुझे पुनः तैनाती के बिना परिचालन नियंत्रण की आवश्यकता है" + -घटनाओं या लागत आयोजनों के दौरान टीमों को त्वरित रनटाइम परिवर्तन की आवश्यकता होती है। +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**ओम्नीरूट इसे कैसे हल करता है:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- कॉम्बो सक्रियण को सीधे एमसीपी डैशबोर्ड से स्विच करें -- पूर्व-निर्धारित पॉलिसी पैक से लचीलापन प्रोफ़ाइल लागू करें -- उसी ऑपरेशन पैनल से सर्किट ब्रेकर स्थिति को रीसेट करें
+**How OmniRoute solves it:** -<विवरण> -<सारांश>🔄 23. "मुझे लाइव ए2ए कार्य जीवनचक्र दृश्यता और रद्दीकरण की आवश्यकता है" +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -जीवनचक्र दृश्यता के बिना, कार्य घटनाओं का परीक्षण करना कठिन हो जाता है। + -**ओम्नीरूट इसे कैसे हल करता है:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- पेजिनेशन के साथ राज्य/कौशल द्वारा कार्य सूचीकरण/फ़िल्टरिंग -- कार्य मेटाडेटा, घटनाओं और कलाकृतियों पर ड्रिल-डाउन -- पुष्टि के साथ कार्य रद्दीकरण समापन बिंदु और यूआई कार्रवाई
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. -<विवरण> -<सारांश>🌊 24. "मुझे A2A लोड के लिए सक्रिय स्ट्रीम मेट्रिक्स की आवश्यकता है" +**How OmniRoute solves it:** -स्ट्रीमिंग वर्कफ़्लो के लिए समवर्ती और लाइव कनेक्शन में परिचालन अंतर्दृष्टि की आवश्यकता होती है। +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**ओम्नीरूट इसे कैसे हल करता है:** + -- सक्रिय स्ट्रीम काउंटर A2A स्थिति में एकीकृत -- अंतिम कार्य टाइमस्टैम्प और प्रति-राज्य गणना -- वास्तविक समय ऑप्स निगरानी के लिए A2A डैशबोर्ड कार्ड +
+📋 20. "I need auditable MCP tool execution" -<विवरण> -<सारांश>🪪 25. "मुझे ग्राहकों के लिए मानक एजेंट खोज की आवश्यकता है" +When tools mutate config or trigger ops actions, teams need forensic traceability. -बाहरी ग्राहकों और ऑर्केस्ट्रेटर्स को ऑनबोर्डिंग के लिए मशीन-पठनीय मेटाडेटा की आवश्यकता होती है। +**How OmniRoute solves it:** -**ओम्नीरूट इसे कैसे हल करता है:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- एजेंट कार्ड `/.well-known/agent.json` पर प्रदर्शित किया गया -- प्रबंधन यूआई में दिखाई गई क्षमताएं और कौशल -- A2A स्थिति API में स्वचालन के लिए खोज मेटाडेटा शामिल है
+ -<विवरण> -<सारांश>🧭 26. "मुझे उत्पाद यूएक्स में प्रोटोकॉल खोज योग्यता की आवश्यकता है" +
+🔐 21. "I need scoped MCP permissions per integration" -यदि उपयोगकर्ता प्रोटोकॉल सतहों की खोज नहीं कर पाते हैं, तो अपनाने और समर्थन की गुणवत्ता में गिरावट आती है। +Different clients should have least-privilege access to tool categories. -**ओम्नीरूट इसे कैसे हल करता है:** +**How OmniRoute solves it:** -- प्रॉक्सी, एमसीपी, ए2ए और एपीआई एंडपॉइंट के लिए टैब के साथ समेकित**एंडपॉइंट**पेज -- एमसीपी और ए2ए के लिए इनलाइन सेवा स्थिति टॉगल (ऑनलाइन/ऑफ़लाइन)। -- सिंहावलोकन से लेकर समर्पित प्रबंधन टैब तक के लिंक
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling -<विवरण> -<सारांश>🧪 27. "मुझे वास्तविक ग्राहकों के साथ एंड-टू-एंड प्रोटोकॉल सत्यापन की आवश्यकता है" + -रिलीज़ से पहले प्रोटोकॉल संगतता को सत्यापित करने के लिए मॉक परीक्षण पर्याप्त नहीं हैं। +
+⚙️ 22. "I need operational controls without redeploying" -**ओम्नीरूट इसे कैसे हल करता है:** +Teams need quick runtime changes during incidents or cost events. -- E2E सुइट जो ऐप को बूट करता है और वास्तविक MCP SDK क्लाइंट ट्रांसपोर्ट का उपयोग करता है -- A2A क्लाइंट खोज, भेजने, स्ट्रीम करने, प्राप्त करने और प्रवाह को रद्द करने के लिए परीक्षण करता है -- एमसीपी ऑडिट और ए2ए कार्य एपीआई के खिलाफ दावों की क्रॉस-चेक करें
+**How OmniRoute solves it:** -<विवरण> -<सारांश>📡 28. "मुझे सभी इंटरफेस में एकीकृत अवलोकन की आवश्यकता है" +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -प्रोटोकॉल द्वारा अवलोकनशीलता को विभाजित करने से ब्लाइंड स्पॉट और लंबा एमटीटीआर बनता है। + -**ओम्नीरूट इसे कैसे हल करता है:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- एक उत्पाद में एकीकृत डैशबोर्ड/लॉग/एनालिटिक्स -- स्वास्थ्य + ऑडिट + ओपनएआई, एमसीपी और ए2ए परतों में टेलीमेट्री अनुरोध -- स्थिति और स्वचालन के लिए परिचालन एपीआई
+Without lifecycle visibility, task incidents become hard to triage. -<विवरण> -<सारांश>💼 29. "मुझे प्रॉक्सी + टूल्स + एजेंट ऑर्केस्ट्रेशन के लिए एक रनटाइम की आवश्यकता है" +**How OmniRoute solves it:** -कई अलग-अलग सेवाएँ चलाने से परिचालन लागत और विफलता मोड बढ़ जाते हैं। +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**ओम्नीरूट इसे कैसे हल करता है:** + -- OpenAI-संगत प्रॉक्सी, MCP सर्वर और A2A सर्वर एक स्टैक में -- साझा प्रमाणीकरण, लचीलापन, डेटा भंडारण और अवलोकन क्षमता -- सभी संपर्क सतहों पर सुसंगत नीति मॉडल +
+🌊 24. "I need active stream metrics for A2A load" -<विवरण> -<सारांश>🚀 30. "मुझे ग्लू-कोड फैलाव के बिना एजेंटिक वर्कफ़्लो भेजने की आवश्यकता है" +Streaming workflows require operational insight into concurrency and live connections. -कई तदर्थ सेवाओं और स्क्रिप्ट्स को सिलाई करते समय टीमों की गति कम हो जाती है। +**How OmniRoute solves it:** -**ओम्नीरूट इसे कैसे हल करता है:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- ग्राहकों और एजेंटों के लिए एकीकृत समापन बिंदु रणनीति -- अंतर्निहित प्रोटोकॉल प्रबंधन यूआई और धूम्रपान सत्यापन पथ -- उत्पादन के लिए तैयार नींव (सुरक्षा, लॉगिंग, लचीलापन, बैकअप)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**प्लेबुक ए: सशुल्क सदस्यता + सस्ता बैकअप अधिकतम करें**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -607,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**प्लेबुक बी: शून्य-लागत कोडिंग स्टैक**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**प्लेबुक सी: 24/7 हमेशा चालू फ़ॉलबैक श्रृंखला**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -630,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**प्लेबुक डी: एजेंट एमसीपी + ए2ए के साथ काम करता है**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost ->**$0/माह**पर मिनटों में AI कोडिंग सेटअप करें। इन मुफ़्त खातों को कनेक्ट करें और अंतर्निहित**फ़्री स्टैक**कॉम्बो का उपयोग करें। +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| कदम | कार्रवाई | प्रदाता अनलॉक | +| Step | Action | Providers Unlocked | | ---- | -------------------------------------------------- | ------------------------------------------------------------------ | -| 1 | कनेक्ट**किरो**(AWS बिल्डर आईडी OAuth) | क्लाउड सॉनेट 4.5, हाइकु 4.5 -**असीमित**| -| 2 | कनेक्ट करें**Qoder**(Google OAuth) | किमी-के2-सोच, क्वेन3-कोडर-प्लस, डीपसीक-आर1... —**असीमित**| -| 3 | कनेक्ट**क्वेन**(डिवाइस कोड) | qwen3-कोडर-प्लस, qwen3-कोडर-फ़्लैश... —**असीमित**| -| 4 | कनेक्ट**मिथुन सीएलआई**(Google OAuth) | जेमिनी-3-फ़्लैश, जेमिनी-2.5-प्रो —**180K/महीना मुफ़्त**| -| 5 | `/डैशबोर्ड/कॉम्बोस` →**फ्री स्टैक ($0)**टेम्पलेट | सभी मुफ़्त प्रदाताओं को स्वचालित रूप से राउंड-रॉबिन करें | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**किसी भी आईडीई/सीएलआई को यहां इंगित करें:**`http://localhost:20128/v1` · एपीआई कुंजी: `any-string` · हो गया। +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**वैकल्पिक अतिरिक्त कवरेज (निःशुल्क भी):**ग्रोक एपीआई कुंजी (30 आरपीएम मुफ्त), एनवीडिया एनआईएम (40 आरपीएम मुफ्त, 70+ मॉडल), सेरेब्रस (1एम टोकन/दिन), लॉन्गकैट एपीआई कुंजी (50एम टोकन/दिन!), क्लाउडफ्लेयर वर्कर्स एआई (10के न्यूरॉन्स/दिन, 50+ मॉडल)।## त्वरित प्रारंभ +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## त्वरित प्रारंभ ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **pnpm उपयोगकर्ता:**`better-sqlite3` और `@swc/core` के लिए आवश्यक मूल बिल्ड स्क्रिप्ट को सक्षम करने के लिए इंस्टॉल के बाद `pnpm Approve-builds -g` चलाएँ: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > -> ```बैश -> पीएनपीएम इंस्टाल -जी ऑम्नीरूट -> पीएनपीएम अप्रूव-बिल्ड्स -जी # सभी पैकेजों का चयन करें → स्वीकृत करें -> सर्वमार्ग +> ```bash +> pnpm install -g omniroute +> pnpm approve-builds -g # Select all packages → approve +> omniroute > ``` -डैशबोर्ड `http://localhost:20128` पर खुलता है और एपीआई बेस यूआरएल `http://localhost:20128/v1` है। +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| आदेश | विवरण | -| ----------------------- | -------------------------------------------------------------------- | ----------- | -| `सर्वव्यापी` | सर्वर प्रारंभ करें (`पोर्ट=20128`, एपीआई और डैशबोर्ड एक ही पोर्ट पर) | -| `ओम्नीरूटे--पोर्ट 3000` | कैनोनिकल/एपीआई पोर्ट को 3000 | पर सेट करें | -| `omniroute --mcp` | MCP सर्वर (stdio ट्रांसपोर्ट) प्रारंभ करें | -| `omniroute --no-open` | ब्राउज़र को स्वतः न खोलें | -| `omniroute --help` | सहायता दिखाएँ | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -वैकल्पिक स्प्लिट-पोर्ट मोड:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -अधिकांश तैनाती के लिए, आपको केवल इसकी आवश्यकता है: +For most deployments, you only need: -| परिवर्तनीय | डिफ़ॉल्ट | उद्देश्य | -| ---------------------- | -------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600000` | अपस्ट्रीम फ़ेच, छिपे हुए अंडरसी टाइमआउट, टीएलएस फ़िंगरप्रिंट अनुरोध और एपीआई ब्रिज अनुरोध/प्रॉक्सी टाइमआउट के लिए साझा आधार रेखा | -| `STREAM_IDLE_TIMEOUT_MS` | `REQUEST_TIMEOUT_MS` प्राप्त होता है | ओम्नीरूट द्वारा एसएसई स्ट्रीम को निरस्त करने से पहले स्ट्रीमिंग खंडों के बीच अधिकतम अंतर +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -बैकवर्ड संगतता संरक्षित है: मौजूदा `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, और अन्य प्रति-लेयर टाइमआउट संस्करण अभी भी काम करते हैं और साझा बेसलाइन को ओवरराइड करते हैं। +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -यदि आपको बेहतर नियंत्रण की आवश्यकता है तो उन्नत ओवरराइड उपलब्ध हैं:| परिवर्तनीय | डिफ़ॉल्ट | उद्देश्य | -| ------------------------------------------------ | ------------------------------------------------ | ---------------------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | `REQUEST_TIMEOUT_MS` प्राप्त होता है | मुख्य फ़ेच एबॉर्ट सिग्नल द्वारा उपयोग किया गया कुल अपस्ट्रीम अनुरोध टाइमआउट | -| `FETCH_HEADERS_TIMEOUT_MS` | `FETCH_TIMEOUT_MS` विरासत में मिला है | अपस्ट्रीम प्रतिक्रिया हेडर प्राप्त करने के लिए निर्धारित समय सीमा | -| `FETCH_BODY_TIMEOUT_MS` | `FETCH_TIMEOUT_MS` विरासत में मिला है | अपस्ट्रीम बॉडी चंक्स के बीच निर्धारित समय सीमा (`0` इसे अक्षम कर देती है) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | अंडरसी टीसीपी कनेक्ट टाइमआउट | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | अन्य आइडल कीप-अलाइव सॉकेट टाइमआउट | -| `TLS_CLIENT_TIMEOUT_MS` | `FETCH_TIMEOUT_MS` विरासत में मिला है | `wreq-js` | के माध्यम से किए गए टीएलएस फिंगरप्रिंट अनुरोधों के लिए टाइमआउट -| `API_BRIDGE_PROXY_TIMEOUT_MS` | `REQUEST_TIMEOUT_MS` या `30000` प्राप्त होता है | एपीआई पोर्ट से डैशबोर्ड पोर्ट तक `/v1` प्रॉक्सी अग्रेषण के लिए टाइमआउट | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `अधिकतम(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | एपीआई ब्रिज सर्वर पर आने वाले अनुरोध का समय समाप्त | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | एपीआई ब्रिज सर्वर पर इनकमिंग हेडर टाइमआउट | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | एपीआई ब्रिज सर्वर पर कीप-अलाइव टाइमआउट | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | एपीआई ब्रिज सर्वर पर सॉकेट निष्क्रियता टाइमआउट (`0` इसे अक्षम करता है) | +Advanced overrides are available if you need finer control: -यदि आप Nginx, Caddy, Cloudflare, या किसी अन्य रिवर्स प्रॉक्सी के पीछे ओमनीरूट चलाते हैं, तो सुनिश्चित करें कि प्रॉक्सी -टाइमआउट आपके ओमनीरूट स्ट्रीम/फ़ेच टाइमआउट से भी अधिक हैं।### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. डैशबोर्ड → `प्रदाता` खोलें और कम से कम एक प्रदाता (OAuth या API कुंजी) कनेक्ट करें। -2. डैशबोर्ड → `एंडपॉइंट्स` खोलें और एक एपीआई कुंजी बनाएं। -3. (वैकल्पिक) डैशबोर्ड → `कॉम्बोस` खोलें और अपनी फ़ॉलबैक श्रृंखला सेट करें।### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -क्लाउड कोड, कोडेक्स सीएलआई, जेमिनी सीएलआई, कर्सर, क्लाइन, ओपनक्लाव, ओपनकोड और ओपनएआई-संगत एसडीके के साथ काम करता है।### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**एमसीपी (टूल-संचालित संचालन के लिए):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -फिर अपने एमसीपी क्लाइंट को `stdio` पर कनेक्ट करें और टूल का परीक्षण करें जैसे: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (एजेंट-टू-एजेंट वर्कफ़्लो के लिए):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -759,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -यह सुइट चल रहे ऐप के विरुद्ध वास्तविक MCP और A2A क्लाइंट प्रवाह को मान्य करता है।### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -767,13 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` -<विवरण> -<सारांश>शून्य लिनक्स (`xbps-src` टेम्पलेट) +
+Void Linux (`xbps-src` template) -Void Linux उपयोगकर्ताओं के लिए, आप `xbps-src` का उपयोग करके एक मूल पैकेज बना सकते हैं। इस ब्लॉक को `srcpkgs/omniroute/template` के रूप में सहेजें:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -785,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -793,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -869,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -880,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -ओमनीरूट [डॉकर हब](https://hub.docker.com/r/diegosouzapw/omniroute) पर सार्वजनिक डॉकर छवि के रूप में उपलब्ध है। +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**तेज़ भागना:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -890,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**पर्यावरण फ़ाइल के साथ:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**डॉकर कंपोज़ का उपयोग करना:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -डॉकर परिनियोजन के लिए डैशबोर्ड समर्थन में अब `डैशबोर्ड → एंडपॉइंट्स` पर एक-क्लिक**क्लाउडफ्लेयर क्विक टनल**शामिल है। पहला, जरूरत पड़ने पर ही `क्लाउडफ्लेयर` को डाउनलोड करने में सक्षम बनाता है, आपके वर्तमान `/v1` समापन बिंदु पर एक अस्थायी सुरंग शुरू करता है, और उत्पन्न `https://*.trycloudflare.com/v1` यूआरएल को सीधे आपके सामान्य सार्वजनिक यूआरएल के नीचे दिखाता है। +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -टिप्पणियाँ: +Notes: -- क्विक टनल यूआरएल अस्थायी होते हैं और हर पुनरारंभ के बाद बदल जाते हैं। -- ओम्निरूट या कंटेनर पुनरारंभ के बाद त्वरित सुरंगें स्वतः बहाल नहीं होती हैं। आवश्यकता पड़ने पर उन्हें डैशबोर्ड से पुनः सक्षम करें। -- प्रबंधित इंस्टाल वर्तमान में `x64` / `arm64` पर Linux, macOS और Windows का समर्थन करता है। -- प्रबंधित त्वरित सुरंगें प्रतिबंधित कंटेनर वातावरण में शोर वाले क्विक यूडीपी बफर चेतावनियों से बचने के लिए HTTP/2 परिवहन के लिए डिफ़ॉल्ट हैं। यदि आप एक अलग परिवहन चाहते हैं तो `CLOUDFLARED_PROTOCOL=quic` या `auto` सेट करें। -- डॉकर छवियां सिस्टम सीए रूट्स को बंडल करती हैं और उन्हें प्रबंधित `क्लाउडफ्लेयर` में भेजती हैं, जो कंटेनर के अंदर सुरंग बूटस्ट्रैप होने पर टीएलएस ट्रस्ट विफलताओं से बचाती है। -- SQLite वाल मोड में चलता है। `डॉकर स्टॉप` को समाप्त होने की अनुमति दी जानी चाहिए ताकि ओमनीरूट नवीनतम परिवर्तनों को `स्टोरेज.स्क्लाइट` में वापस चेकपॉइंट कर सके। -- बंडल की गई कंपोज़ फ़ाइलें पहले से ही 40s स्टॉप ग्रेस अवधि निर्धारित करती हैं। यदि आप छवि को सीधे चलाते हैं, तो `--स्टॉप-टाइमआउट 40` (या समान) रखें ताकि मैन्युअल स्टॉप शटडाउन क्लीनअप में कटौती न करें। -- यदि आप चाहते हैं कि ओमनीरूट डाउनलोड करने के बजाय मौजूदा बाइनरी का उपयोग करे तो `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` सेट करें। +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**कैडी (HTTPS ऑटो-टीएलएस) के साथ डॉकर कंपोज़ का उपयोग करना:** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -कैडी के स्वचालित एसएसएल प्रावधान का उपयोग करके ओमनीरूट को सुरक्षित रूप से उजागर किया जा सकता है। सुनिश्चित करें कि आपके डोमेन का DNS A रिकॉर्ड आपके सर्वर के आईपी को इंगित करता है।```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` +| Image | Tag | Size | Description | +| ------------------------ | -------- | ------ | --------------------- | +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | -| छवि | टैग | आकार | विवरण | -| ---------------------- | -------- | ------ | ---------------------- | -| `diegosouzapw/omniroute` | `नवीनतम` | ~250एमबी | नवीनतम स्थिर रिलीज़ | -| `diegosouzapw/omniroute` | `1.0.3` | ~250एमबी | वर्तमान संस्करण |--- +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**नया!**ओमनीरूट अब विंडोज, मैकओएस और लिनक्स के लिए**नेटिव डेस्कटॉप एप्लिकेशन**के रूप में उपलब्ध है। +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -ओम्नीरूट को एक स्टैंडअलोन डेस्कटॉप ऐप के रूप में चलाएं - स्थानीय मॉडलों के लिए कोई टर्मिनल, कोई ब्राउज़र, कोई इंटरनेट आवश्यक नहीं है। इलेक्ट्रॉन-आधारित ऐप में शामिल हैं: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**नेटिव विंडो**- सिस्टम ट्रे एकीकरण के साथ समर्पित ऐप विंडो -- 🔄**ऑटो-स्टार्ट**— सिस्टम लॉगिन पर ओमनीरूट लॉन्च करें -- 🔔**मूल सूचनाएं**- कोटा समाप्त होने या प्रदाता समस्याओं के लिए अलर्ट प्राप्त करें -- ⚡**वन-क्लिक इंस्टॉल**- एनएसआईएस (विंडोज़), डीएमजी (मैकओएस), ऐपइमेज (लिनक्स) -- 🌐**ऑफ़लाइन मोड**— बंडल सर्वर के साथ पूरी तरह ऑफ़लाइन काम करता है### त्वरित प्रारंभ +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### त्वरित प्रारंभ ```bash # Development mode @@ -979,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -न्यूनतम होने पर, ओमनीरूट त्वरित क्रियाओं के साथ आपके सिस्टम ट्रे में रहता है: +When minimized, OmniRoute lives in your system tray with quick actions: -- डैशबोर्ड खोलें -- सर्वर पोर्ट बदलें -- आवेदन छोड़ें +- Open dashboard +- Change server port +- Quit application -📖 पूर्ण दस्तावेज: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| टियर | प्रदाता | लागत | कोटा रीसेट | के लिए सर्वश्रेष्ठ | -| ----------------- | ---------------------------- | -------------------------------- | ---------------------- | -------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 सदस्यता** | क्लाउड कोड (प्रो) | $20/माह | 5 घंटे + साप्ताहिक | पहले ही सदस्यता ले ली है | -| | कोडेक्स (प्लस/प्रो) | $20-200/महीना | 5 घंटे + साप्ताहिक | OpenAI उपयोगकर्ता | -| | जेमिनी सीएलआई | **मुफ़्त** | 180K/माह + 1K/दिन | सब लोग! | -| | गिटहब कोपायलट | $10-19/माह | मासिक | GitHub उपयोगकर्ता | -| **🔑एपीआई कुंजी** | एनवीडिया एनआईएम | **मुफ़्त**(हमेशा के लिए देव) | ~40 आरपीएम | 70+ खुले मॉडल | -| | सेरेब्रस | **मुफ़्त**(1 मिलियन टोकन/दिन) | 60K टीपीएम / 30 आरपीएम | दुनिया का सबसे तेज़ | -| | ग्रोक | **मुफ़्त**(30 आरपीएम) | 14.4K आरपीडी | अल्ट्रा-फास्ट लामा/जेम्मा | -| | डीपसीक V3.2 | $0.27/$1.10 प्रति 1 मिलियन | कोई नहीं | सर्वोत्तम मूल्य/गुणवत्ता तर्क | -| | xAI ग्रोक-4 फास्ट | **$0.20/$0.50 प्रति 1 मिलियन**🆕 | कोई नहीं | सबसे तेज़ + टूल कॉलिंग, अल्ट्रालो | -| | xAI ग्रोक-4 (मानक) | $0.20/$1.50 प्रति 1 मिलियन 🆕 | कोई नहीं | xAI से रीज़निंग फ्लैगशिप | -| | मिस्ट्रल | नि:शुल्क परीक्षण + सशुल्क | दर सीमित | यूरोपीय एआई | -| | ओपनराउटर | भुगतान-प्रति-उपयोग | कोई नहीं | 100+ मॉडल कुल मिलाकर। | -| **💰सस्ता** | GLM-5 (Z.AI के माध्यम से) 🆕 | $0.5/1 मिलियन | प्रतिदिन सुबह 10 बजे | 128K आउटपुट, नवीनतम फ्लैगशिप | -| | जीएलएम-4.7 | $0.6/1 मिलियन | प्रतिदिन सुबह 10 बजे | बजट बैकअप | -| | मिनीमैक्स एम2.5 🆕 | $0.3/1M इनपुट | 5 घंटे की रोलिंग | तर्क + एजेंटिक कार्य | -| | मिनीमैक्स एम2.1 | $0.2/1 मिलियन | 5 घंटे की रोलिंग | सबसे सस्ता विकल्प | -| | किमी K2.5 (मूनशॉट एपीआई) 🆕 | भुगतान-प्रति-उपयोग | कोई नहीं | डायरेक्ट मूनशॉट एपीआई एक्सेस | -| | किमी K2 | $9/महीना फ्लैट | 10एम टोकन/माह | अनुमानित लागत | -| **🆓 मुफ़्त** | कोडर | **$0** | असीमित | 5 मॉडल असीमित | -| | क्वेन | **$0** | असीमित | 4 मॉडल असीमित | -| | किरो | **$0** | असीमित | क्लाउड सॉनेट/हाइकू (एडब्ल्यूएस बिल्डर) | -| | लॉन्गकैट फ्लैश-लाइट 🆕 | **$0**(50 मिलियन टोकन/दिन 🔥) | 1 आरपीएस | पृथ्वी पर सबसे बड़ा मुफ़्त कोटा | -| | परागण एआई 🆕 | **$0**(कोई कुंजी आवश्यक नहीं) | 1 अनुरोध/15s | जीपीटी-5, क्लाउड, डीपसीक, लामा 4 | -| | क्लाउडफ्लेयर वर्कर्स एआई 🆕 | **$0**(10K न्यूरॉन्स/दिन) | ~150 सम्मान/दिन | 50+ मॉडल, वैश्विक बढ़त | -| | स्केलवे एआई 🆕 | **$0**(कुल 1 मिलियन टोकन) | दर सीमित | ईयू/जीडीपीआर, क्वेन3 235बी, लामा 70बी | > 🆕**नए मॉडल जोड़े गए (मार्च 2026):**$0.20/$0.50/M पर ग्रोक-4 फास्ट परिवार (1143ms पर बेंचमार्क - जेमिनी 2.5 फ्लैश से 30% तेज), 128K आउटपुट के साथ Z.AI के माध्यम से GLM-5, मिनीमैक्स M2.5 रीजनिंग, डीपसीक V3.2 अद्यतन मूल्य निर्धारण, मूनशॉट डायरेक्ट एपीआई के माध्यम से किमी K2.5। | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 $0 कॉम्बो स्टैक - पूर्ण निःशुल्क सेटअप:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**शून्य लागत. कोडिंग कभी बंद नहीं होती।**इसे एक ओमनीरूट कॉम्बो के रूप में कॉन्फ़िगर करें और सभी फ़ॉलबैक स्वचालित रूप से होते हैं - कोई मैन्युअल स्विचिंग नहीं।--- +--- --- ## 🆓 Free Models — What You Actually Get -> नीचे दिए गए सभी मॉडल**शून्य क्रेडिट कार्ड की आवश्यकता के साथ 100% निःशुल्क**हैं। जब एक कोटा समाप्त हो जाता है तो ओमनीरूट उनके बीच ऑटो-रूट करता है - उन सभी को एक अटूट $0 कॉम्बो के लिए संयोजित करें।### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| मॉडल | उपसर्ग | सीमा | दर सीमा | -| ------------------- | ------ | ----------------- | ---------------------- | -| `क्लाउड-सॉनेट-4.5` | `क्र/` |**असीमित**| कोई रिपोर्ट नहीं की गई दैनिक सीमा | -| `क्लाउड-हाइकु-4.5` | `क्र/` |**असीमित**| कोई रिपोर्ट नहीं की गई दैनिक सीमा | -| `क्लाउड-ओपस-4.6` | `क्र/` |**असीमित**| किरो के माध्यम से नवीनतम रचना |### 🟢 QODER MODELS (Free PAT via qodercli) +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) -| मॉडल | उपसर्ग | सीमा | दर सीमा | -| ------------------ | ------ | ----------------- | --------------- | -| `किमी-के2-सोच` | `अगर/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं | -| `क्वेन3-कोडर-प्लस` | `अगर/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं | -| `डीपसीक-आर1` | `अगर/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं | -| `मिनीमैक्स-एम2.1` | `अगर/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं | -| `किमी-के2` | `अगर/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं | +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | --------------------- | +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -> अनुशंसित कनेक्शन विधि:**पर्सनल एक्सेस टोकन + `qodercli`**। ब्राउज़र OAuth है -> प्रयोगात्मक और डिफ़ॉल्ट रूप से अक्षम जब तक कि `QODER_OAUTH_*` पर्यावरण चर कॉन्फ़िगर नहीं किए जाते।### 🟡 QWEN MODELS (Device Code Auth) +### 🟢 QODER MODELS (Free PAT via qodercli) -| मॉडल | उपसर्ग | सीमा | दर सीमा | -| ------------------- | ------ | ----------------- | ------------------- | -| `क्वेन3-कोडर-प्लस` | `qw/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं | -| `क्वेन3-कोडर-फ़्लैश` | `qw/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं | -| `qwen3-कोडर-नेक्स्ट` | `qw/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं | -| `विज़न-मॉडल` | `qw/` |**असीमित**| मल्टीमॉडल (चित्र) |### 🟣 GEMINI CLI (Google OAuth) +| Model | Prefix | Limit | Rate Limit | +| ------------------ | ------ | ------------- | --------------- | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -| मॉडल | उपसर्ग | सीमा | दर सीमा | -| ---------------------- | ------ | -------------------------------- | ----------------- | -| `मिथुन-3-फ़्लैश-पूर्वावलोकन` | `जीसी/` |**180K टोकन/माह**+ 1K/दिन | मासिक रीसेट | -| `मिथुन-2.5-प्रो` | `जीसी/` | 180K/माह (साझा पूल) | उच्च गुणवत्ता |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| टियर | दैनिक सीमा | दर सीमा | नोट्स | -| ---------- | ----------- | ----------- | ---------------------------------------------------------------- | -| मुफ़्त (देव) | कोई टोकन सीमा नहीं |**~40 आरपीएम**| 70+ मॉडल; 2025 के मध्य में शुद्ध दर सीमा में परिवर्तन | +### 🟡 QWEN MODELS (Device Code Auth) -लोकप्रिय मुफ्त मॉडल: `मूनशोताई/किमी-के2.5` (किमी के2.5), `जेड-एआई/जीएलएम4.7` (जीएलएम 4.7), `डीपसीक-एआई/डीपसीक-वी3.2` (डीपसीक वी3.2), `एनवीडिया/ल्लामा-3.3-70बी-इंस्ट्रक्ट`, `डीपसीक/डीपसीक-आर1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | ------------------- | +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| टियर | दैनिक सीमा | दर सीमा | नोट्स | -| ---- | ----------------- | ---------------- | ------------------------------------------------ | -| मुफ़्त |**1 मिलियन टोकन/दिन**| 60K टीपीएम / 30 आरपीएम | दुनिया का सबसे तेज़ एलएलएम अनुमान; प्रतिदिन रीसेट होता है | +### 🟣 GEMINI CLI (Google OAuth) -निःशुल्क उपलब्ध: `लामा-3.3-70बी`, `लामा-3.1-8बी`, `डीपसीक-आर1-डिस्टिल-लामा-70बी`### 🔴 GROQ (Free API Key — console.groq.com) +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -| टियर | दैनिक सीमा | दर सीमा | नोट्स | -| ---- | ----------------- | ---------------- | ------------------------------------------------ | -| मुफ़्त |**14.4K आरपीडी**| प्रति मॉडल 30 आरपीएम | कोई क्रेडिट कार्ड नहीं; 429 सीमा पर, शुल्क नहीं | +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) -निःशुल्क उपलब्ध: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +| Tier | Daily Limit | Rate Limit | Notes | +| ---------- | ------------ | ----------- | ------------------------------------------------------ | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -| मॉडल | उपसर्ग | दैनिक निःशुल्क कोटा | नोट्स | -| -------------------------------- | ------ | ----------------- | ---------------------- | -| `लॉन्गकैट-फ्लैश-लाइट` | `एलसी/` |**50M टोकन**💥 | अब तक का सबसे बड़ा मुफ़्त कोटा | -| 'लॉन्गकैट-फ्लैश-चैट' | `एलसी/` | 500K टोकन | मल्टी-टर्न चैट | -| 'लॉन्गकैट-फ्लैश-थिंकिंग' | `एलसी/` | 500K टोकन | तर्क/सीओटी | -| `लॉन्गकैट-फ्लैश-थिंकिंग-2601` | `एलसी/` | 500K टोकन | जनवरी 2026 संस्करण | -| `लॉन्गकैट-फ्लैश-ओमनी-2603` | `एलसी/` | 500K टोकन | मल्टीमॉडल | +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -> सार्वजनिक बीटा में रहते हुए 100% निःशुल्क। ईमेल या फोन से [longcat.chat](https://longcat.chat) पर साइन अप करें। प्रतिदिन 00:00 UTC पर रीसेट होता है।### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -| मॉडल | उपसर्ग | दर सीमा | पीछे प्रदाता | +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | + +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` + +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ------------- | ---------------- | ----------------------------------------- | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | + +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` + +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 + +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | + +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. + +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| 'ओपनाई' | `पोल/` | 1 अनुरोध/15s | जीपीटी-5 | -| 'क्लाउड' | `पोल/` | 1 अनुरोध/15s | एंथ्रोपिक क्लाउड | -| 'मिथुन' | `पोल/` | 1 अनुरोध/15s | गूगल जेमिनी | -| 'डीपसीक' | `पोल/` | 1 अनुरोध/15s | डीपसीक वी3 | -| 'लामा' | `पोल/` | 1 अनुरोध/15s | मेटा लामा 4 स्काउट | -| 'मिस्ट्रल' | `पोल/` | 1 अनुरोध/15s | मिस्ट्रल एआई | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**शून्य घर्षण:**कोई साइनअप नहीं, कोई एपीआई कुंजी नहीं। खाली कुंजी फ़ील्ड के साथ परागण प्रदाता जोड़ें और यह तुरंत काम करता है।### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| टियर | दैनिक न्यूरॉन्स | समतुल्य उपयोग | नोट्स | -| ---- | ----------------- | ------------------------------------------------ | ---------------------- | -| मुफ़्त |**10,000**| ~150 एलएलएम सम्मान / 500s ऑडियो / 15K एंबेड | वैश्विक बढ़त, 50+ मॉडल | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 -लोकप्रिय मुफ़्त मॉडल: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (मुफ़्त ऑडियो!), `@cf/qwen/qwen2.5-coder-15b-instruct` +| Tier | Daily Neurons | Equivalent Usage | Notes | +| ---- | ------------- | --------------------------------------- | ----------------------- | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -> [dash.cloudflare.com](https://dash.cloudflare.com) से एपीआई टोकन + खाता आईडी की आवश्यकता है। प्रदाता सेटिंग्स में खाता आईडी संग्रहीत करें।### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -| टियर | मुफ़्त कोटा | स्थान | नोट्स | -| ---- | ----------------- | ----------- | -------------------------------------- | -| मुफ़्त |**1M टोकन**| 🇫🇷पेरिस, ईयू | सीमा के भीतर किसी क्रेडिट कार्ड की आवश्यकता नहीं | +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -निःशुल्क उपलब्ध: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `depseek-v3-0324` +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 -> ईयू/जीडीपीआर के अनुरूप। [console.scaleway.com](https://console.scaleway.com) पर एपीआई कुंजी प्राप्त करें। +| Tier | Free Quota | Location | Notes | +| ---- | ------------- | ------------ | ----------------------------------- | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | ->**💡 परम निःशुल्क स्टैक (11 प्रदाता, $0 हमेशा के लिए):** +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` + +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). + +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> किरो (केआर/) → क्लाउड सॉनेट/हाइकु अनलिमिटेड -> कोडर (यदि/) → किमी-के2-सोच, क्वेन3-कोडर-प्लस, डीपसीक-आर1 अनलिमिटेड -> लॉन्गकैट लाइट (एलसी/) → लॉन्गकैट-फ्लैश-लाइट - 50एम टोकन/दिन 🔥 -> परागण (पोल/) → जीपीटी-5, क्लाउड, डीपसीक, लामा 4 - किसी कुंजी की आवश्यकता नहीं -> क्वेन (qw/) → क्वेन3-कोडर मॉडल असीमित -> मिथुन (मिथुन /) → मिथुन 2.5 फ्लैश - 1,500 अनुरोध/दिन निःशुल्क -> क्लाउडफ्लेयर एआई (सीएफ/) → 50+ मॉडल - 10K न्यूरॉन्स/दिन -> स्केलवे (scw/) → Qwen3 235B, Llama 70B — 1M मुफ़्त टोकन (EU) -> ग्रोक (ग्रोक/) → लामा/जेम्मा - 14.4K अनुरोध/दिन अल्ट्रा-फास्ट -> एनवीडिया एनआईएम (एनवीडिया/) → 70+ खुले मॉडल - 40 आरपीएम हमेशा के लिए -> सेरेब्रस (सेरेब्रस/) → लामा/क्वेन दुनिया का सबसे तेज़ - 1 मिलियन टोकन/दिन -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` ->**$0**के लिए किसी भी ऑडियो/वीडियो को ट्रांसक्राइब करें - डीपग्राम $200 मुफ़्त, असेंबलीएआई $50 फ़ॉलबैक, ग्रोक व्हिस्पर असीमित आपातकालीन बैकअप के साथ अग्रणी है। +## 🎙️ Free Transcription Combo -| प्रदाता | मुफ़्त क्रेडिट | सर्वश्रेष्ठ मॉडल | दर सीमा | -| ----------------- | ---------------------- | ------------------------------------------------ | -------------------------------- | -| 🟢**दीपग्राम**|**$200 निःशुल्क**(साइनअप) | `नोवा-3` — सर्वोत्तम सटीकता, 30+ भाषाएँ | निःशुल्क क्रेडिट पर कोई आरपीएम सीमा नहीं | -| 🔵**असेंबलीएआई**|**$50 निःशुल्क**(साइनअप) | `यूनिवर्सल-3-प्रो` — अध्याय, भावना, पीआईआई | निःशुल्क क्रेडिट पर कोई आरपीएम सीमा नहीं | -| 🔴**ग्रोक**|**हमेशा के लिए मुफ़्त**| `व्हिस्पर-लार्ज-v3` - ओपनएआई व्हिस्पर | 30 आरपीएम (दर सीमित) | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. -**`/डैशबोर्ड/कॉम्बोस` में सुझाया गया कॉम्बो:**``` +| Provider | Free Credits | Best Model | Rate Limit | +| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | + +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -फिर `/डैशबोर्ड/मीडिया` →**ट्रांसक्रिप्शन**टैब में: कोई भी ऑडियो या वीडियो फ़ाइल अपलोड करें → अपना कॉम्बो एंडपॉइंट चुनें → समर्थित प्रारूपों में ट्रांसक्रिप्शन प्राप्त करें।## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -ओमनीरूट v2.0 को केवल एक रिले प्रॉक्सी नहीं, बल्कि एक ऑपरेशनल प्लेटफॉर्म के रूप में बनाया गया है।### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| फ़ीचर | यह क्या करता है | -| ---------------------------------- | -------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**ग्रोक-4 फास्ट फ़ैमिली** | $0.20/$0.50/M पर xAI मॉडल - बेंचमार्क 1143ms (जेमिनी 2.5 फ्लैश से 30% तेज) | -| 🧠**Z.AI के माध्यम से GLM-5** | 128K आउटपुट संदर्भ, $0.5/1M - GLM परिवार का नवीनतम फ्लैगशिप | -| 🔮**मिनीमैक्स एम2.5** | $0.30/1 मिलियन पर रीज़निंग + एजेंटिक कार्य - एम2.1 से महत्वपूर्ण उन्नयन | -| 🎯**टूलकॉलिंग फ़्लैग प्रति मॉडल** | रजिस्ट्री में प्रति-मॉडल `टूलकॉलिंग: सही/गलत` - ऑटोकॉम्बो गैर-टूल-सक्षम मॉडल को छोड़ देता है | -| 🌍**बहुभाषी आशय का पता लगाना** | ऑटोकॉम्बो स्कोरिंग में पीटी/जेडएच/ईएस/एआर कीवर्ड - गैर-अंग्रेजी सामग्री के लिए बेहतर मॉडल चयन | -| 📊**बेंचमार्क-प्रेरित फ़ॉलबैक** | लाइव अनुरोधों से वास्तविक p95 विलंबता कॉम्बो स्कोरिंग फ़ीड करती है - ऑटोकॉम्बो वास्तविक डेटा से सीखता है | -| 🔁**डुप्लीकेशन का अनुरोध** | कंटेंट-हैश आधारित डिडअप विंडो - मल्टी-एजेंट सुरक्षित, डुप्लिकेट शुल्क को रोकता है | -| 🔌**प्लग करने योग्य राउटर रणनीति** | एक्स्टेंसिबल `राउटरस्ट्रैटेजी` इंटरफ़ेस - प्लगइन्स के रूप में कस्टम रूटिंग लॉजिक जोड़ें | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| फ़ीचर | यह क्या करता है | -| -------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------- | -| 🎮**मॉडल खेल का मैदान** | किसी भी मॉडल का सीधे परीक्षण करने के लिए डैशबोर्ड पेज - प्रदाता/मॉडल/एंडपॉइंट चयनकर्ता, मोनाको संपादक, स्ट्रीमिंग, निरस्त, समय | -| 🔏**सीएलआई फ़िंगरप्रिंट मिलान** | मूल सीएलआई हस्ताक्षरों से मिलान करने के लिए प्रति-प्रदाता हेडर/बॉडी ऑर्डरिंग - सेटिंग्स> सुरक्षा में प्रति प्रदाता टॉगल करें।**आपका प्रॉक्सी आईपी संरक्षित है** | -| 🤝**एसीपी सपोर्ट (एजेंट क्लाइंट प्रोटोकॉल)** | सीएलआई एजेंट खोज (कोडेक्स, क्लाउड, गूज़, जेमिनी सीएलआई, ओपनक्लॉ + 9 अधिक), प्रोसेस स्पॉनर, `/api/acp/agents` एंडपॉइंट | -| 🤖**एसीपी एजेंट्स डैशबोर्ड** | डीबग › एजेंट पेज - किसी भी सीएलआई टूल के लिए इंस्टॉल स्थिति, संस्करण, कस्टम एजेंट फॉर्म के साथ 14 एजेंटों का ग्रिड।**ओपनकोड**उपयोगकर्ताओं को एक "डाउनलोड ओपनकोड.जेसन" बटन मिलता है जो सभी उपलब्ध मॉडलों के साथ उपयोग के लिए तैयार कॉन्फ़िगरेशन को स्वतः उत्पन्न करता है। | -| 🔧**कस्टम मॉडल `एपीआईफॉर्मेट` रूटिंग** | `apiFormat: "प्रतिक्रियाएं"` के साथ कस्टम मॉडल अब प्रतिक्रिया एपीआई अनुवादक पर सही ढंग से रूट करते हैं | -| 🏢**कोडेक्स कार्यक्षेत्र अलगाव** | प्रति ईमेल एकाधिक कोडेक्स कार्यस्थान - OAuth कार्यस्थान आईडी द्वारा कनेक्शन को सही ढंग से अलग करता है | -| 🔄**इलेक्ट्रॉन ऑटो-अपडेट** | डेस्कटॉप ऐप अपडेट की जांच करता है + रीस्टार्ट होने पर ऑटो-इंस्टॉल | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| फ़ीचर | यह क्या करता है | -| --------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**एमसीपी सर्वर (25 उपकरण)** | 3 ट्रांसपोर्ट के माध्यम से आईडीई/एजेंट उपकरण: stdio, SSE (`/api/mcp/sse`), स्ट्रीम करने योग्य HTTP (`/api/mcp/stream`)। 18 कोर + 3 मेमोरी + 4 कौशल उपकरण | -| 🤝**A2A सर्वर (JSON-RPC + SSE)** | सिंक और स्ट्रीमिंग प्रवाह के साथ एजेंट-टू-एजेंट कार्य निष्पादन | -| 🧭**समेकित समापन बिंदु पृष्ठ** | एंडपॉइंट प्रॉक्सी, एमसीपी, ए2ए और एपीआई एंडपॉइंट टैब के साथ टैब्ड प्रबंधन पृष्ठ | -| 🎚️**सेवा सक्षम/अक्षम टॉगल** | सेटिंग्स दृढ़ता के साथ एमसीपी और ए2ए के लिए चालू/बंद स्विच (डिफ़ॉल्ट: बंद) | -| 🛰️**एमसीपी रनटाइम हार्टबीट** | वास्तविक प्रक्रिया स्थिति (पीआईडी, अपटाइम, दिल की धड़कन की उम्र, परिवहन, स्कोप मोड) | -| 📋**एमसीपी ऑडिट ट्रेल** | सफलता/असफलता और मुख्य एट्रिब्यूशन के साथ फ़िल्टर करने योग्य ऑडिट लॉग | -| 🔐**एमसीपी स्कोप प्रवर्तन** | नियंत्रित टूल एक्सेस के लिए 10 ग्रैन्युलर स्कोप अनुमतियाँ | -| 📡**A2A कार्य जीवनचक्र प्रबंधन** | कार्यों को सूचीबद्ध करें/फ़िल्टर करें, घटनाओं/कलाकृतियों का निरीक्षण करें, चल रहे कार्यों को रद्द करें | -| 📋**एजेंट कार्ड डिस्कवरी** | क्लाइंट ऑटो-डिस्कवरी के लिए `/.well-known/agent.json` | -| 🧪**प्रोटोकॉल E2E टेस्ट हार्नेस** | वास्तविक MCP SDK + A2A क्लाइंट `test:protocols:e2e` | में प्रवाहित होता है | -| ⚙️**परिचालन नियंत्रण** | कॉम्बो स्विच करें, लचीलापन प्रोफ़ाइल लागू करें, एक नियंत्रण सतह से ब्रेकर रीसेट करें | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| फ़ीचर | यह क्या करता है | -| -------------------------------- | ------------------------------------------------------------------------- | ----------------------- | -| 🎯**स्मार्ट 4-टियर फ़ॉलबैक** | ऑटो-रूट: सदस्यता → एपीआई कुंजी → सस्ता → मुफ़्त | -| 📊**वास्तविक समय कोटा ट्रैकिंग** | लाइव टोकन गिनती + प्रति प्रदाता रीसेट उलटी गिनती | -| 🔄**प्रारूप अनुवाद** | OpenAI ↔ क्लाउड ↔ जेमिनी ↔ स्कीमा-सुरक्षित रूपांतरण के साथ प्रतिक्रियाएँ | -| 👥**मल्टी-अकाउंट सपोर्ट** | बुद्धिमान चयन के साथ प्रति प्रदाता एकाधिक खाते | -| 🔄**ऑटो टोकन रिफ्रेश** | OAuth टोकन पुनः प्रयास के साथ स्वचालित रूप से ताज़ा हो जाते हैं | -| 🎨**कस्टम कॉम्बो** | 9 संतुलन रणनीतियाँ + फ़ॉलबैक श्रृंखला नियंत्रण | -| 🌐**वाइल्डकार्ड राउटर** | `प्रदाता/*` डायनेमिक रूटिंग | -| 🧠**सोच बजट नियंत्रण** | पासथ्रू, ऑटो, कस्टम और अनुकूली तर्क सीमाएँ | -| 🔀**मॉडल उपनाम** | बिल्ट-इन + कस्टम मॉडल अलियासिंग और माइग्रेशन सुरक्षा | -| ⚡**पृष्ठभूमि का क्षरण** | कम प्राथमिकता वाले पृष्ठभूमि कार्यों को सस्ते मॉडल पर रूट करें | -| 🧪**टास्क-अवेयर स्मार्ट रूटिंग** | सामग्री प्रकार (कोडिंग/विज़न/विश्लेषण/सारांशीकरण) द्वारा स्वतः-चयन मॉडल | -| 🔄**A2A एजेंट वर्कफ़्लोज़** | स्टेटफुल मल्टी-स्टेप एजेंट निष्पादन के लिए नियतात्मक एफएसएम ऑर्केस्ट्रेटर | -| 🔀**अनुकूली रूटिंग** | टोकन वॉल्यूम और शीघ्र जटिलता के आधार पर गतिशील रणनीति ओवरराइड | -| 🎲**प्रदाता विविधता** | शैनन एन्ट्रापी स्कोरिंग संतुलन ऑटो-कॉम्बो ट्रैफ़िक वितरण | -| 💬**सिस्टम प्रॉम्प्ट इंजेक्शन** | वैश्विक व्यवहार नियंत्रण लगातार लागू | -| 📄**प्रतिक्रियाएं एपीआई संगतता** | कोडेक्स और उन्नत एजेंटिक वर्कफ़्लोज़ के लिए पूर्ण `/v1/responses` समर्थन | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| फ़ीचर | यह क्या करता है | -| --------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**छवि निर्माण** | `/v1/images/जेनरेशन` क्लाउड और स्थानीय बैकएंड के साथ | -| 📐**एंबेडिंग** | खोज और RAG पाइपलाइनों के लिए `/v1/embeddings` | -| 🎤**ऑडियो ट्रांस्क्रिप्शन** | `/v1/ऑडियो/ट्रांसक्रिप्शन` - 7 प्रदाता (डीपग्राम नोवा 3, असेंबलीएआई, ग्रोक व्हिस्पर, हगिंगफेस, इलेवनलैब्स, ओपनएआई, एज़्योर), ऑटो-लैंग्वेज डिटेक्शन, एमपी4/एमपी3/डब्ल्यूएवी सपोर्ट | -| 🔊**टेक्स्ट-टू-स्पीच** | `/v1/ऑडियो/स्पीच` - सही त्रुटि संदेशों के साथ 10 प्रदाता (इलेवनलैब्स, ओपनएआई, डीपग्राम, कार्टेसिया, प्लेएचटी, हगिंगफेस, एनवीडिया एनआईएम, इनवर्ल्ड, कोक्वी, टोर्टोइज़) | -| 🎬**वीडियो जेनरेशन** | `/v1/वीडियो/पीढ़ी` (ComfyUI + SD WebUI वर्कफ़्लोज़) | -| 🎵**संगीत पीढ़ी** | `/v1/संगीत/पीढ़ी` (ComfyUI वर्कफ़्लोज़) | -| 🛡️**संयम** | `/v1/मॉडरेशन` सुरक्षा जांच | -| 🔀**पुनर्रैंकिंग** | प्रासंगिकता स्कोरिंग के लिए `/v1/rerank` | -| 🔍**वेब खोज**🆕 | `/v1/search` - 5 प्रदाता (सर्पर, ब्रेव, पर्प्लेक्सिटी, एक्सा, टैविली), 6,500+ मुफ़्त/माह, ऑटो-फ़ेलओवर, कैश | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| फ़ीचर | यह क्या करता है | -| ------------------------------------------ | ----------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------- | -| 🔌**सर्किट तोड़ने वाले** | प्रति-मॉडल यात्रा/सीमा नियंत्रण के साथ पुनर्प्राप्ति | -| 🎯**एंडपॉइंट-अवेयर मॉडल** | कस्टम मॉडल समर्थित एंडपॉइंट + एपीआई प्रारूप की घोषणा करते हैं | -| 🛡️**एंटी-थंडरिंग झुंड** | पुनः प्रयास/दर घटनाओं पर म्यूटेक्स + सेमाफोर सुरक्षा | -| 🧠**सिमेंटिक + सिग्नेचर कैश** | दो कैश परतों के साथ लागत/विलंबता में कमी | -| ⚡**निष्क्रियता का अनुरोध** | डुप्लिकेट सुरक्षा विंडो | -| 🔒**टीएलएस फ़िंगरप्रिंट स्पूफिंग** | ब्राउज़र जैसा टीएलएस फ़िंगरप्रिंट -**बॉट डिटेक्शन और अकाउंट फ़्लैगिंग को कम करता है** | -| 🔏**सीएलआई फ़िंगरप्रिंट मिलान** | मूल सीएलआई अनुरोध हस्ताक्षरों से मेल खाता है -**प्रॉक्सी आईपी को संरक्षित करते हुए प्रतिबंध जोखिम को कम करता है** | -| 🌐**आईपी फ़िल्टरिंग** | उजागर तैनाती के लिए अनुमति सूची/अवरुद्ध सूची नियंत्रण | -| 📊**संपादन योग्य दर सीमाएँ** | दृढ़ता के साथ कॉन्फ़िगर करने योग्य वैश्विक/प्रदाता-स्तर की सीमाएं | -| 📉**सौम्य पतन** | मल्टी-लेयर क्षमता फ़ॉलबैक कोर गेटवे ऑपरेशंस की सुरक्षा करती है | -| 📜**कॉन्फिग ऑडिट ट्रेल** | डिफ-आधारित परिवर्तन ट्रैकिंग सरल रोलबैक के साथ परिचालन बहाव को रोकती है | -| ⏳**प्रदाता स्वास्थ्य सिंक** | प्राधिकरण विफलताओं से पहले ट्रिगरिंग अलर्ट सक्रिय टोकन समाप्ति निगरानी | -| 🚪**प्रतिबंधित खातों को स्वतः अक्षम करें** | ऑपरेशनल सर्किट ब्रेकर स्वचालित रूप से स्थायी रूप से ब्लॉक किए गए टोकन खातों को सील कर देता है | -| 🔑**एपीआई कुंजी प्रबंधन + स्कोपिंग** | सुरक्षित कुंजी जारी करना/रोटेशन और मॉडल/प्रदाता नियंत्रण | -| 👁️**स्कोप्ड एपीआई कुंजी का खुलासा**🆕 | `ALLOW_API_KEY_REVEAL` | के माध्यम से एपीआई कुंजियों की ऑप्ट-इन पुनर्प्राप्ति | -| 🛡️**संरक्षित `/मॉडल`** | मॉडल कैटलॉग के लिए वैकल्पिक प्रमाणीकरण गेटिंग और प्रदाता छिपाना | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| फ़ीचर | यह क्या करता है | -| ---------------------------------- | ----------------------------------------------------------------------- | ---------------------------- | -| 📝**अनुरोध + प्रॉक्सी लॉगिंग** | पूर्ण अनुरोध/प्रतिक्रिया और प्रॉक्सी लॉगिंग | -| 📉**स्ट्रीम किए गए विस्तृत लॉग**🆕 | एसएसई पेलोड स्ट्रीम को यूआई में साफ-सुथरा तरीके से पुनर्निर्माण करता है | -| 📋**एकीकृत लॉग डैशबोर्ड** | एक पृष्ठ में अनुरोध, प्रॉक्सी, ऑडिट और कंसोल दृश्य | -| 🔍**टेलीमेट्री के लिए अनुरोध** | p50/p95/p99 विलंबता और अनुरोध अनुरेखण | -| 🏥**स्वास्थ्य डैशबोर्ड** | अपटाइम, ब्रेकर स्थिति, लॉकआउट, कैश आँकड़े | -| 💰**लागत ट्रैकिंग** | बजट नियंत्रण और प्रति-मॉडल मूल्य निर्धारण दृश्यता | -| 📈**एनालिटिक्स विज़ुअलाइज़ेशन** | मॉडल/प्रदाता उपयोग अंतर्दृष्टि और रुझान दृश्य | -| 🧪**मूल्यांकन ढाँचा** | विन्यास योग्य मिलान रणनीतियों के साथ गोल्डन सेट परीक्षण | -| 📡**लाइव डायग्नोस्टिक्स**🆕 | सटीक कॉम्बो लाइव परीक्षण के लिए सिमेंटिक कैश बाईपास | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| फ़ीचर | यह क्या करता है | -| --------------------------------- | ----------------------------------------------------------------------------------- | ------------------------------------ | -| 🌐**कहीं भी तैनात करें** | लोकलहोस्ट, वीपीएस, डॉकर, क्लाउड वातावरण | -| 🚇**क्लाउडफ्लेयर टनल**🆕 | डैशबोर्ड से एक-क्लिक त्वरित सुरंग एकीकरण | -| 🔑**एपीआई कुंजी मॉडल फ़िल्टरिंग** | मूल /v1/मॉडल प्रतिक्रिया निर्दिष्ट बियरर संदर्भ भूमिकाओं के माध्यम से फ़िल्टर की गई | -| ⚡**स्मार्ट कैश बायपास** | कॉन्फ़िगर करने योग्य टीटीएल अनुमान और फ़ोर्स्ड रीफ़ेच नियंत्रण | -| 🔄**बैकअप/पुनर्स्थापना** | निर्यात/आयात और आपदा पुनर्प्राप्ति प्रवाह | -| 🧙**ऑनबोर्डिंग विज़ार्ड** | फर्स्ट-रन गाइडेड सेटअप | -| 🔧**सीएलआई टूल्स डैशबोर्ड** | लोकप्रिय कोडिंग टूल के लिए एक-क्लिक सेटअप | -| 🎮**मॉडल खेल का मैदान** | डैशबोर्ड से किसी भी प्रदाता/मॉडल/एंडपॉइंट का परीक्षण करें | -| 🔏**सीएलआई फ़िंगरप्रिंट टॉगल** | सेटिंग्स > सुरक्षा | में प्रति प्रदाता फ़िंगरप्रिंट मिलान | -| 🌐**i18n (30 भाषाएँ)** | पूर्ण डैशबोर्ड + आरटीएल कवरेज के साथ डॉक्स भाषा समर्थन | -| 🧹**सभी मॉडल साफ़ करें** | प्रदाता विवरण में एक-क्लिक मॉडल सूची समाशोधन | -| 👁️**साइडबार नियंत्रण**🆕 | उपस्थिति सेटिंग्स से घटकों और एकीकरणों को छुपाएं | -| 📋**मुद्दा टेम्पलेट** | बग और सुविधाओं के लिए मानकीकृत GitHub टेम्पलेट | -| 📂**कस्टम डेटा निर्देशिका** | भंडारण स्थान के लिए `DATA_DIR` ओवरराइड | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1292,103 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -जब कोटा, दर, या स्वास्थ्य विफल हो जाता है, तो ओमनीरूट मैन्युअल स्विचिंग के बिना स्वचालित रूप से अगले उम्मीदवार के पास चला जाता है।#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A UI और डॉक्स में खोजने योग्य हैं (छिपा हुआ नहीं) -- प्रोटोकॉल स्थिति एपीआई लाइव परिचालन डेटा को उजागर करते हैं (`/api/mcp/*`, `/api/a2a/*`) -- डैशबोर्ड में दिन-2 ऑप्स के लिए क्रियाएं शामिल हैं (कॉम्बो टॉगल, ब्रेकर रीसेट, कार्य रद्द करना)#### Translator + validation workflow +#### Protocol management that is visible and operable -अनुवादक क्षेत्र में शामिल हैं: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**खेल का मैदान**: परिवर्तन जांच का अनुरोध करें -**चैट परीक्षक**: पूर्ण अनुरोध/प्रतिक्रिया राउंड-ट्रिप -**टेस्ट बेंच**: एक बार में कई मामले -**लाइव मॉनिटर**: वास्तविक समय यातायात दृश्य +#### Translator + validation workflow -साथ ही `npm run test:protocols:e2e` के माध्यम से वास्तविक ग्राहकों के साथ प्रोटोकॉल सत्यापन। +The Translator area includes: -> 📖**[एमसीपी सर्वर रीडमी](ओपन-एसएसई/एमसीपी-सर्वर/रीडमी.एमडी)**- टूल संदर्भ, आईडीई कॉन्फ़िगरेशन और क्लाइंट उदाहरण +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A सर्वर README](src/lib/a2a/README.md)**- कौशल, JSON-RPC विधियाँ, स्ट्रीमिंग, और कार्य जीवनचक्र## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -ओमनीरूट में गोल्डन सेट के मुकाबले एलएलएम प्रतिक्रिया गुणवत्ता का परीक्षण करने के लिए एक अंतर्निहित मूल्यांकन ढांचा शामिल है। डैशबोर्ड में**एनालिटिक्स → इवेल्स**के माध्यम से इसे एक्सेस करें।### Built-in Golden Set +## 🧪 Evaluations (Evals) -प्री-लोडेड "ओम्नीरूट गोल्डन सेट" में इसके लिए परीक्षण मामले शामिल हैं: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- नमस्ते, गणित, भूगोल, कोड जनरेशन -- JSON प्रारूप अनुपालन, अनुवाद, मार्कडाउन पीढ़ी -- सुरक्षा इनकार (हानिकारक सामग्री), गिनती, बूलियन तर्क### Evaluation Strategies +### Built-in Golden Set -| रणनीति | विवरण | उदाहरण | -| ---------- | ------------------------------------------------- | ------------------------------ | --- | -| 'सटीक' | आउटपुट बिल्कुल मेल खाना चाहिए | `"4"` | -| 'शामिल है' | आउटपुट में सबस्ट्रिंग (केस-असंवेदनशील) होना चाहिए | `"पेरिस"` | -| 'रेगेक्स' | आउटपुट रेगेक्स पैटर्न से मेल खाना चाहिए | `"1.*2.*3"` | -| `कस्टम` | कस्टम जेएस फ़ंक्शन सही/गलत लौटाता है | `(आउटपुट) => आउटपुट.लेंथ > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) -<विवरण> -<सारांश>🧩 एमसीपी सेटअप (मॉडल संदर्भ प्रोटोकॉल) +
+🧩 MCP Setup (Model Context Protocol) -stdio मोड में MCP ट्रांसपोर्ट प्रारंभ करें:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -अनुशंसित सत्यापन प्रवाह: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. अपने MCP क्लाइंट को stdio से कनेक्ट करें। -2. `omniroute_get_health` चलाएँ। -3. `omniroute_list_combos` चलाएँ। -4. दिल की धड़कन, गतिविधि और ऑडिट की पुष्टि करने के लिए `/डैशबोर्ड/एमसीपी` खोलें। +Useful APIs for automation: -स्वचालन के लिए उपयोगी एपीआई: +- `GET /api/mcp/status` +- `GET /api/mcp/tools` +- `GET /api/mcp/audit` +- `GET /api/mcp/audit/stats` -- `प्राप्त करें /एपीआई/एमसीपी/स्थिति` -- `प्राप्त करें /एपीआई/एमसीपी/टूल्स` -- `प्राप्त करें /एपीआई/एमसीपी/ऑडिट` -- `प्राप्त करें /api/mcp/ऑडिट/आँकड़े`
+ -<विवरण> -<सारांश>🤝 A2A सेटअप (एजेंट2एजेंट) +
+🤝 A2A Setup (Agent2Agent) -एजेंट का पता लगाएं:```bash +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -एक कार्य भेजें:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` +Manage lifecycle: -जीवनचक्र प्रबंधित करें: - -- `प्राप्त करें /api/a2a/status` -- `प्राप्त करें /api/a2a/कार्य` -- `प्राप्त करें /api/a2a/tasks/:id` +- `GET /api/a2a/status` +- `GET /api/a2a/tasks` +- `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -परिचालन यूआई: +Operational UI: -- कार्य/स्थिति/स्ट्रीम अवलोकन और धूम्रपान क्रियाओं के लिए `/डैशबोर्ड/ए2ए`
+- `/dashboard/a2a` for task/state/stream observability and smoke actions -<विवरण> -<सारांश>🧪 एंड-टू-एंड प्रोटोकॉल सत्यापन + -वास्तविक ग्राहकों के साथ दोनों प्रोटोकॉल मान्य करें:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -यह सत्यापित करता है: +This verifies: -- एमसीपी एसडीके क्लाइंट कनेक्ट/लिस्ट/कॉल -- A2A खोज/भेजें/स्ट्रीम/प्राप्त करें/रद्द करें -- एमसीपी ऑडिट और ए2ए कार्य प्रबंधन एपीआई में डेटा को क्रॉस-चेक करें
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs -<विवरण> -<सारांश>💳 सदस्यता प्रदाता### Claude Code (Pro/Max) + + +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1401,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**प्रो टिप:**जटिल कार्यों के लिए ओपस और गति के लिए सॉनेट का उपयोग करें। ओमनीरूट प्रति मॉडल कोटा ट्रैक करता है!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1415,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -प्रत्येक कोडेक्स खाते में अब `डैशबोर्ड -> प्रदाता` में नीति टॉगल हैं: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5 घंटे` (चालू/बंद): 5 घंटे की विंडो सीमा नीति लागू करें। -- `साप्ताहिक` (चालू/बंद): साप्ताहिक विंडो सीमा नीति लागू करें। -- थ्रेसहोल्ड व्यवहार: जब एक सक्षम विंडो >=90% उपयोग तक पहुंच जाती है, तो वह खाता छोड़ दिया जाता है। -- रोटेशन व्यवहार: ओमनीरूट स्वचालित रूप से अगले पात्र कोडेक्स खाते पर रूट करता है। -- रीसेट व्यवहार: जब प्रदाता का `resetAt` समय बीत जाता है, तो खाता स्वचालित रूप से फिर से पात्र हो जाता है। +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -परिदृश्य: +Scenarios: -- `5 घंटे चालू` + `साप्ताहिक चालू`: जब कोई भी विंडो सीमा तक पहुंचती है तो खाता छोड़ दिया जाता है। -- `5 घंटे की छूट` + `साप्ताहिक चालू`: केवल साप्ताहिक उपयोग ही खाते को ब्लॉक कर सकता है। -- `5 घंटे चालू` + `साप्ताहिक बंद`: केवल 5 घंटे का उपयोग ही खाते को ब्लॉक कर सकता है। -- `resetAt` पारित: खाता स्वचालित रूप से रोटेशन में पुनः प्रवेश करता है (कोई मैन्युअल पुनः सक्षम नहीं)।### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1440,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**सर्वोत्तम मूल्य:**विशाल निःशुल्क स्तर! सशुल्क स्तरों से पहले इसका उपयोग करें।### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1455,71 +1662,91 @@ Models:
-<विवरण> -<सारांश>🔑 एपीआई कुंजी प्रदाता### NVIDIA NIM (FREE developer access — 70+ models) +
+🔑 API Key Providers -1. साइन अप करें: [build.nvidia.com](https://build.nvidia.com) -2. निःशुल्क एपीआई कुंजी प्राप्त करें (1000 अनुमान क्रेडिट शामिल) -3. डैशबोर्ड → प्रदाता जोड़ें → एनवीडिया एनआईएम: - - एपीआई कुंजी: `nvapi-your-key` +### NVIDIA NIM (FREE developer access — 70+ models) -**मॉडल:**`एनवीडिया/लामा-3.3-70बी-इंस्ट्रक्ट`, `एनवीडिया/मिस्ट्रल-7बी-इंस्ट्रक्ट`, और 50+ अधिक +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**प्रो टिप:**ओपनएआई-संगत एपीआई - ओमनीरूट के प्रारूप अनुवाद के साथ सहजता से काम करता है!### DeepSeek +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -1. साइन अप करें: [प्लेटफ़ॉर्म.डीपसीक.कॉम](https://platform.डीपसीक.कॉम) -2. एपीआई कुंजी प्राप्त करें -3. डैशबोर्ड → प्रदाता जोड़ें → डीपसीक +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -**मॉडल:**`डीपसीक/डीपसीक-चैट`, `डीपसीक/डीपसीक-कोडर`### Groq (Free Tier Available!) +### DeepSeek -1. साइन अप करें: [console.groq.com](https://console.groq.com) -2. एपीआई कुंजी प्राप्त करें (फ्री टियर शामिल) -3. डैशबोर्ड → प्रदाता जोड़ें → ग्रोक +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -**मॉडल:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**प्रो टिप:**अल्ट्रा-फास्ट अनुमान - वास्तविक समय कोडिंग के लिए सर्वोत्तम!### OpenRouter (100+ Models) +### Groq (Free Tier Available!) -1. साइन अप करें: [openrouter.ai](https://openrouter.ai) -2. एपीआई कुंजी प्राप्त करें -3. डैशबोर्ड → प्रदाता जोड़ें → ओपनराउटर +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -**मॉडल:**एक ही एपीआई कुंजी के माध्यम से सभी प्रमुख प्रदाताओं से 100+ मॉडल तक पहुंचें। +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**डैशबोर्ड व्यवहार:**ओपनराउटर मॉडल**उपलब्ध मॉडल**से प्रबंधित किए जाते हैं। मैन्युअल ऐड, आयात और ऑटो-सिंक सभी एक ही सूची को अपडेट करते हैं।
+**Pro Tip:** Ultra-fast inference — best for real-time coding! -<विवरण> -<सारांश>💰 सस्ते प्रदाता (बैकअप)### GLM-4.7 (Daily reset, $0.6/1M) +### OpenRouter (100+ Models) -1. साइन अप करें: [झिपु एआई](https://open.bigmodel.cn/) -2. कोडिंग योजना से एपीआई कुंजी प्राप्त करें -3. डैशबोर्ड → एपीआई कुंजी जोड़ें: - - प्रदाता: `glm` - - एपीआई कुंजी: `आपकी-कुंजी` +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -**उपयोग:**`glm/glm-4.7` +**Models:** Access 100+ models from all major providers through a single API key. -**प्रो टिप:**कोडिंग प्लान 1/7 लागत पर 3× कोटा प्रदान करता है! प्रतिदिन सुबह 10:00 बजे रीसेट करें।### MiniMax M2.1 (5h reset, $0.20/1M) +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -1. साइन अप करें: [मिनीमैक्स](https://www.minimax.io/) -2. एपीआई कुंजी प्राप्त करें -3. डैशबोर्ड → एपीआई कुंजी जोड़ें + -**उपयोग करें:**`मिनीमैक्स/मिनीमैक्स-एम2.1` +
+💰 Cheap Providers (Backup) -**प्रो टिप:**लंबे संदर्भ के लिए सबसे सस्ता विकल्प (1M टोकन)!### Kimi K2 ($9/month flat) +### GLM-4.7 (Daily reset, $0.6/1M) -1. सदस्यता लें: [मूनशॉट एआई](https://platform.moonshot.ai/) -2. एपीआई कुंजी प्राप्त करें -3. डैशबोर्ड → एपीआई कुंजी जोड़ें +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**उपयोग करें:**`किमी/किमी-नवीनतम` +**Use:** `glm/glm-4.7` -**प्रो टिप:**10एम टोकन के लिए निश्चित $9/माह = $0.90/1एम प्रभावी लागत!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. -<विवरण> -<सारांश>🆓 मुफ़्त प्रदाता (आपातकालीन बैकअप)### Qoder (5 FREE models via OAuth) +### MiniMax M2.1 (5h reset, $0.20/1M) + +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `minimax/MiniMax-M2.1` + +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1560,8 +1787,10 @@ Models:
-<विवरण> -<सारांश>🎨 कॉम्बो बनाएं### Example 1: Maximize Subscription → Cheap Backup +
+🎨 Create Combos + +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1589,8 +1818,10 @@ Cost: $0 forever!
-<विवरण> -<सारांश>🔧 सीएलआई एकीकरण### Cursor IDE +
+🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1601,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -एक-क्लिक कॉन्फ़िगरेशन के लिए डैशबोर्ड में**सीएलआई टूल्स**पृष्ठ का उपयोग करें, या `~/.claude/settings.json` को मैन्युअल रूप से संपादित करें।### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1612,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**विकल्प 1 - डैशबोर्ड (अनुशंसित):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**विकल्प 2 - मैनुअल:**`~/.openclaw/openclaw.json` संपादित करें:```json +```json { "models": { "providers": { @@ -1629,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **ध्यान दें:**ओपनक्लाव केवल स्थानीय ओमनीरूट के साथ काम करता है। IPv6 रिज़ॉल्यूशन समस्याओं से बचने के लिए `लोकलहोस्ट` के बजाय `127.0.0.1` का उपयोग करें।### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1643,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**चरण 1:**एक कस्टम प्रदाता के रूप में ओम्निरूट जोड़ें:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**चरण 2:**अपने प्रोजेक्ट रूट में `opencode.json` बनाएं/संपादित करें:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1669,118 +1909,130 @@ opencode } } } -```` +``` -**चरण 3:**ओपनकोड में मॉडल का चयन करें:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**टिप:**अपने ओमनीरूट `/v1/models` एंडपॉइंट में उपलब्ध किसी भी मॉडल को `मॉडल` अनुभाग में जोड़ें। अपने ओमनीरूट डैशबोर्ड से `प्रदाता/मॉडल-आईडी` प्रारूप का उपयोग करें।
+ --- ## समस्या निवारण -<विवरण> -<सारांश>समस्या निवारण मार्गदर्शिका का विस्तार करने के लिए क्लिक करें +
+Click to expand troubleshooting guide -**"भाषा मॉडल ने संदेश प्रदान नहीं किया"** +**"Language model did not provide messages"** -- प्रदाता कोटा समाप्त → डैशबोर्ड कोटा ट्रैकर की जाँच करें -- समाधान: कॉम्बो फ़ॉलबैक का उपयोग करें या सस्ते स्तर पर स्विच करें +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**दर सीमित करना** +**Rate limiting** -- सदस्यता कोटा ख़त्म → GLM/MiniMax पर फ़ॉलबैक -- कॉम्बो जोड़ें: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**OAuth टोकन समाप्त हो गया** +**OAuth token expired** -- ओम्निरूट द्वारा स्वतः ताज़ा -- यदि समस्या बनी रहती है: डैशबोर्ड → प्रदाता → पुनः कनेक्ट करें +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**उच्च लागत** +**High costs** -- डैशबोर्ड → लागत में उपयोग के आँकड़े जाँचें -- प्राथमिक मॉडल को जीएलएम/मिनीमैक्स पर स्विच करें -- गैर-महत्वपूर्ण कार्यों के लिए फ्री टियर (मिथुन सीएलआई, क्यूडर) का उपयोग करें +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**डैशबोर्ड/एपीआई पोर्ट गलत हैं** +**Dashboard/API ports are wrong** -- `पोर्ट` कैनोनिकल बेस पोर्ट है (और डिफ़ॉल्ट रूप से एपीआई पोर्ट) -- `API_PORT` केवल OpenAI-संगत API श्रोता को ओवरराइड करता है -- `DASHBOARD_PORT` केवल डैशबोर्ड/Next.js श्रोता को ओवरराइड करता है -- अपने डैशबोर्ड/सार्वजनिक URL पर `NEXT_PUBLIC_BASE_URL` सेट करें (OAuth कॉलबैक के लिए) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**क्लाउड सिंक त्रुटियाँ** +**Cloud sync errors** -- अपने चल रहे उदाहरण के लिए `BASE_URL` बिंदुओं को सत्यापित करें -- अपने अपेक्षित क्लाउड एंडपॉइंट पर `CLOUD_URL` बिंदुओं को सत्यापित करें -- `NEXT_PUBLIC_*` मानों को सर्वर-साइड मानों के साथ संरेखित रखें +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**पहला लॉगिन काम नहीं कर रहा** +**First login not working** -- `.env` में `INITIAL_PASSWORD` जांचें -- यदि सेट नहीं है, तो फ़ॉलबैक पासवर्ड `123456` है +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**कोई अनुरोध लॉग नहीं** +**No request logs** -- अनुरोध कलाकृतियों को प्रति अनुरोध एक JSON फ़ाइल के रूप में `DATA_DIR/call_logs/` पर लिखा जाता है -- यदि आपको विस्तृत प्रति-स्टेज पेलोड की आवश्यकता है तो डैशबोर्ड → लॉग → अनुरोध लॉग से पाइपलाइन कैप्चर सक्षम करें -- यदि आप `logs/application/app.log` में एप्लिकेशन कंसोल लॉग भी चाहते हैं तो `APP_LOG_TO_FILE=true` सेट करें -- `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, और `CALL_LOG_MAX_ENTRIES` को आवश्यकतानुसार समायोजित करें +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**कनेक्शन परीक्षण OpenAI-संगत प्रदाताओं के लिए "अमान्य" दिखाता है** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- कई प्रदाता `/मॉडल` समापन बिंदु को उजागर नहीं करते हैं -- ओमनीरूट v1.0.6+ में चैट पूर्णता के माध्यम से फ़ॉलबैक सत्यापन शामिल है -- सुनिश्चित करें कि आधार URL में `/v1` प्रत्यय शामिल है### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix - - +### 🔐 OAuth on a Remote Server ->**⚠️ वीपीएस, डॉकर या किसी रिमोट सर्वर पर ओमनीरूट चलाने वाले उपयोगकर्ताओं के लिए महत्वपूर्ण**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? + + -**एंटीग्रेविटी**और**जेमिनी सीएलआई**प्रदाता**Google OAuth 2.0**का उपयोग करते हैं। Google को ऐप के Google क्लाउड कंसोल में पूर्व-पंजीकृत यूआरआई में से एक से सटीक मिलान करने के लिए OAuth प्रवाह में `redirect_uri` की आवश्यकता है। +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -ओम्निरूट में बंडल किए गए OAuth क्रेडेंशियल**केवल `लोकलहोस्ट`* के लिए पंजीकृत हैं। जब आप किसी दूरस्थ सर्वर (उदाहरण के लिए `https://omniroute.myserver.com`) पर ओमनीरूट एक्सेस करते हैं, तो Google प्रमाणीकरण को अस्वीकार कर देता है:``` +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? + +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -आपको अपने सर्वर के यूआरआई के साथ Google क्लाउड कंसोल में एक**OAuth 2.0 क्लाइंट आईडी**बनाना होगा।#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Google क्लाउड कंसोल खोलें** +#### Step-by-step -यहां जाएं: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. एक नया OAuth 2.0 क्लाइंट आईडी बनाएं** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) --**"+ क्रेडेंशियल बनाएं"**→**"OAuth क्लाइंट आईडी"**पर क्लिक करें +**2. Create a new OAuth 2.0 Client ID** -- एप्लिकेशन प्रकार:**"वेब एप्लिकेशन"** -- नाम: कुछ भी जो आपको पसंद हो (जैसे `ओम्नीरूट रिमोट`) +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -**3. अधिकृत रीडायरेक्ट यूआरआई जोड़ें** +**3. Add Authorized Redirect URIs** -**"अधिकृत रीडायरेक्ट यूआरआई"**फ़ील्ड में, जोड़ें:``` +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> `your-server.com` को अपने सर्वर के डोमेन या आईपी से बदलें (यदि आवश्यक हो तो पोर्ट शामिल करें, उदाहरण के लिए `http://45.33.32.156:20128/callback`)। +**4. Save and copy the credentials** -**4. क्रेडेंशियल सहेजें और कॉपी करें** +After creating, Google will show the **Client ID** and **Client Secret**. -बनाने के बाद, Google**क्लाइंट आईडी**और**क्लाइंट सीक्रेट**दिखाएगा। +**5. Set environment variables** -**5. पर्यावरण चर सेट करें** +In your `.env` (or Docker environment variables): -आपके `.env` (या डॉकर पर्यावरण चर) में:```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1789,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. ओम्निरूट को पुनरारंभ करें**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. पुनः कनेक्ट करने का प्रयास करें** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -डैशबोर्ड → प्रदाता → एंटीग्रेविटी (या जेमिनी सीएलआई) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google अब `https://your-server.com/callback` पर सही ढंग से रीडायरेक्ट करेगा।--- +--- #### Temporary workaround (without custom credentials) -यदि आप अभी अपना स्वयं का क्रेडेंशियल सेट नहीं करना चाहते हैं, तो आप अभी भी**मैन्युअल यूआरएल प्रवाह**का उपयोग कर सकते हैं: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. ओमनीरूट Google प्राधिकरण URL खोलता है -2. अधिकृत करने के बाद, Google `लोकलहोस्ट` पर रीडायरेक्ट करने का प्रयास करता है (जो रिमोट सर्वर पर विफल रहता है) -3.**अपने ब्राउज़र के एड्रेस बार से पूरा यूआरएल कॉपी करें**(भले ही पेज लोड न हो) -4. उस यूआरएल को ओमनीरूट कनेक्शन मोडल में दिखाए गए फ़ील्ड में पेस्ट करें -5.**"कनेक्ट"**पर क्लिक करें +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> यह काम करता है क्योंकि यूआरएल में प्राधिकरण कोड इस बात पर ध्यान दिए बिना मान्य है कि रीडायरेक्ट पेज लोड किया गया है या नहीं।--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. -<विवरण> -<सारांश>🇧🇷 पुर्तगाली भाषा में वर्साओ#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -प्रमाणित करने के लिए**एंटीग्रेविटी**और**मिथुन सीएलआई**का उपयोग करें**Google OAuth 2.0**का उपयोग करें। Google को एक `redirect_uri` का उपयोग करना चाहिए जो बिना किसी प्रवाह के OAuth सेजा**exatamente**का उपयोग करता है और आपको Google क्लाउड कंसोल के लिए यूआरआई डाउनलोड करने की आवश्यकता है। +
+🇧🇷 Versão em Português -जैसा कि OAuth का क्रेडेंशियल है, कोई ओम्निरूट एस्टाओ कैडस्ट्रास नहीं है**'लोकलहोस्ट'**के लिए एपेनास। एक सर्विडोर रिमोट (उदा: `https://omniroute.meuservidor.com`) पर ओम्निरूट का उपयोग कैसे करें, या Google एक ऑटेंटिका कॉम को पुनः प्राप्त करता है:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -आपका सटीक विवरण**OAuth 2.0 क्लाइंट आईडी**आपके सर्वर पर यूआरआई के साथ Google क्लाउड कंसोल नहीं है।#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Google क्लाउड कंसोल तक पहुंच** +#### Passo a passo -अब्राहम: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Acesse o Google Cloud Console** -**2. नया OAuth 2.0 क्लाइंट आईडी देखें** +Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- उन्हें क्लिक करें**"+ क्रेडेंशियल बनाएं"**→**"OAuth क्लाइंट आईडी"** -- आवेदन टिप:**"वेब एप्लिकेशन"** -- नोम: एस्कोल्हा क्वाल्कर नोम (उदा: `ओम्नीरूट रिमोट`) +**2. Crie um novo OAuth 2.0 Client ID** -**3. अधिकृत रीडायरेक्ट यूआरआई के रूप में एडिकियोन** +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -कोई शिकायत नहीं**"अधिकृत रीडायरेक्ट यूआरआई"**, आदि:``` +**3. Adicione as Authorized Redirect URIs** + +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> स्थानापन्न `seu-servidor.com` अपने आईपी को अपने सर्वर पर रखें (इसमें एक आवश्यक पोर्ट भी शामिल है, उदाहरण के लिए: `http://45.33.32.156:20128/callback`)। +**4. Salve e copie as credenciais** -**4. साख के रूप में सहेजें और कॉपी करें** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -एपोस क्रियर, Google द्वारा**क्लाइंट आईडी**और**क्लाइंट सीक्रेट**। +**5. Configure as variáveis de ambiente** -**5. परिवेश परिवर्तन**के रूप में कॉन्फ़िगर करें +No seu `.env` (ou nas variáveis de ambiente do Docker): -कोई सेउ `.env` (आप डॉकर के परिवेश को कैसे बदलते हैं):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1868,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. ओम्निरूट का नवीनीकरण**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute +``` -```` +**7. Tente conectar novamente** -**7. नए सिरे से संपर्क करें** +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -डैशबोर्ड → प्रदाता → एंटीग्रेविटी (या जेमिनी सीएलआई) → OAuth +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. -यदि आप `https://seu-servidor.com/callback` और एक प्रामाणिक कार्य के लिए Google पुनर्निर्देशन करते हैं।--- +--- #### Workaround temporário (sem configurar credenciais próprias) -यदि आप पहले से ही उचित क्रेडेंशियल प्राप्त नहीं करना चाहते हैं, तो आपके लिए फ्लक्सो का उपयोग करना संभव है**यूआरएल का मैनुअल**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. Google पर स्वचालित URL का उपयोग करके ओम्निरूट का उपयोग करें -2. आप स्वचालित रूप से काम कर सकते हैं, या Google `लोकलहोस्ट` को पुनः प्राप्त कर सकता है (यदि कोई सर्वर रिमोट नहीं है) -3.**एक यूआरएल को पूरा कॉपी करें**अपने ब्राउजर से दोबारा डाउनलोड करें (मुझे लगता है कि एक पेज अभी भी उपलब्ध है) -4. ओम्निरूट से जुड़ने के लिए कोई भी यूआरएल नहीं है -5. उन्हें क्लिक करें**"कनेक्ट"** +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) +4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute +5. Clique em **"Connect"** -> यह समाधान यूआरएल को स्वचालित रूप से डाउनलोड करने के लिए काम कर रहा है और आपके द्वारा किए गए रीडायरेक्ट को स्वतंत्र रूप से वैध बनाता है।
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1906,64 +2171,73 @@ docker restart omniroute ## 🛠️ Tech Stack -<विवरण> -<सारांश>तकनीकी स्टैक विवरण का विस्तार करने के लिए क्लिक करें +
+Click to expand tech stack details --**रनटाइम**: Node.js 18-22 LTS (⚠️ Node.js 24+**समर्थित नहीं**है - `better-sqlite3` मूल बायनेरिज़ असंगत हैं) --**भाषा**: टाइपस्क्रिप्ट 5.9 -**100% टाइपस्क्रिप्ट**`src/` और `open-sse/` में (v2.0 के बाद से कोर मॉड्यूल में शून्य `कोई भी`) --**फ्रेमवर्क**: नेक्स्ट.जेएस 16 + रिएक्ट 19 + टेलविंड सीएसएस 4 --**डेटाबेस**: LowDB (JSON) + SQLite (डोमेन स्थिति + प्रॉक्सी लॉग + MCP ऑडिट + रूटिंग निर्णय) --**स्कीमा**: ज़ॉड (एमसीपी टूल I/O सत्यापन, एपीआई अनुबंध) --**प्रोटोकॉल**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**स्ट्रीमिंग**: सर्वर-भेजे गए इवेंट (एसएसई) --**प्रामाणिक**: OAuth 2.0 (PKCE) + JWT + API कुंजियाँ + MCP स्कोप्ड प्राधिकरण --**परीक्षण**: Node.js टेस्ट रनर + विटेस्ट (यूनिट, एकीकरण, E2E सहित 900+ परीक्षण) --**सीआई/सीडी**: गिटहब क्रियाएँ (ऑटो एनपीएम प्रकाशन + रिलीज पर डॉकर हब) --**वेबसाइट**: [omniroute.online](https://omniroute.online) --**पैकेज**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**डॉकर**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**लचीलापन**: सर्किट ब्रेकर, एक्सपोनेंशियल बैकऑफ़, एंटी-थंडरिंग झुंड, टीएलएस स्पूफिंग, ऑटो-कॉम्बो सेल्फ-हीलिंग
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## दस्तावेज़ -| दस्तावेज़ | विवरण | -| ------------------------------------------------ | ---------------------------------------------------------------- | -| [उपयोगकर्ता गाइड](docs/USER_GUIDE.md) | प्रदाता, कॉम्बो, सीएलआई एकीकरण, तैनाती | -| [एपीआई संदर्भ](docs/API_REFERENCE.md) | उदाहरण सहित सभी समापन बिंदु | -| [एमसीपी सर्वर](ओपन-एसएसई/एमसीपी-सर्वर/रीडमी.एमडी) | 16 एमसीपी उपकरण, आईडीई कॉन्फ़िगरेशन, पायथन/टीएस/गो क्लाइंट | -| [ए2ए सर्वर](src/lib/a2a/README.md) | JSON-RPC 2.0 प्रोटोकॉल, कौशल, स्ट्रीमिंग, कार्य प्रबंधन | -| [ऑटो-कॉम्बो इंजन](docs/auto-combo.md) | 6-कारक स्कोरिंग, मोड पैक, स्व-उपचार | -| [समस्या निवारण](docs/TROUBLESHOOTING.md) | सामान्य समस्याएँ एवं समाधान | -| [आर्किटेक्चर](docs/ARCHITECTURE.md) | सिस्टम आर्किटेक्चर और आंतरिक | -| [योगदान](CONTRIBUTING.md) | विकास सेटअप और दिशानिर्देश | -| [OpenAPI Spec](docs/openapi.yaml) | ओपनएपीआई 3.0 विशिष्टता | -| [सुरक्षा नीति](सुरक्षा.एमडी) | भेद्यता रिपोर्टिंग और सुरक्षा प्रथाएं | -| [VM परिनियोजन](docs/VM_DEPLOYMENT_GUIDE.md) | संपूर्ण गाइड: VM + nginx + Cloudflare सेटअप | -| [फीचर्स गैलरी](docs/FEATURES.md) | स्क्रीनशॉट के साथ विजुअल डैशबोर्ड टूर | -| [रिलीज़ चेकलिस्ट](docs/RELEASE_CHECKLIST.md) | प्री-रिलीज़ सत्यापन चरण |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -ओम्निरूट ने कई विकास चरणों में**210+ सुविधाओं की योजना बनाई है**। यहां प्रमुख क्षेत्र हैं: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| श्रेणी | नियोजित विशेषताएं | हाइलाइट्स | -| -------------------------------- | ---------------- | -------------------------------------------------------------------------------------------------- | -| 🧠**रूटिंग और इंटेलिजेंस**| 25+ | न्यूनतम-विलंबता रूटिंग, टैग-आधारित रूटिंग, कोटा प्रीफ़्लाइट, पी2सी खाता चयन | -| 🔒**सुरक्षा एवं अनुपालन**| 20+ | एसएसआरएफ हार्डनिंग, क्रेडेंशियल क्लोकिंग, प्रति समापन बिंदु दर-सीमा, प्रबंधन कुंजी स्कोपिंग | -| 📊**अवलोकनशीलता**| 15+ | ओपन टेलीमेट्री एकीकरण, वास्तविक समय कोटा निगरानी, ​​प्रति मॉडल लागत ट्रैकिंग | -| 🔄**प्रदाता एकीकरण**| 20+ | डायनेमिक मॉडल रजिस्ट्री, प्रदाता कूलडाउन, मल्टी-अकाउंट कोडेक्स, कोपायलट कोटा पार्सिंग | -| ⚡**प्रदर्शन**| 15+ | दोहरी कैश परत, शीघ्र कैश, प्रतिक्रिया कैश, स्ट्रीमिंग कीपलाइव, बैच एपीआई | -| 🌐**पारिस्थितिकी तंत्र**| 10+ | वेबसॉकेट एपीआई, कॉन्फिग हॉट-रीलोड, वितरित कॉन्फिग स्टोर, वाणिज्यिक मोड |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**ओपनकोड इंटीग्रेशन**- ओपनकोड एआई कोडिंग आईडीई के लिए मूल प्रदाता समर्थन -- 🔗**TRAE एकीकरण**- TRAE AI विकास ढांचे के लिए पूर्ण समर्थन -- 📦**बैच एपीआई**- थोक अनुरोधों के लिए अतुल्यकालिक बैच प्रोसेसिंग -- 🎯**टैग-आधारित रूटिंग**- कस्टम टैग और मेटाडेटा के आधार पर रूट अनुरोध -- 💰**न्यूनतम-लागत रणनीति**— स्वचालित रूप से सबसे सस्ते उपलब्ध प्रदाता का चयन करें +### 🔜 Coming Soon -> 📝 पूर्ण सुविधा विशिष्टताएँ [`docs/new-features/`](docs/new-features/) में उपलब्ध हैं (217 विस्तृत विवरण)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1971,18 +2245,20 @@ docker restart omniroute ### How to Contribute -1. रिपॉजिटरी को फोर्क करें -2. अपनी फीचर शाखा बनाएं (`git checkout -b फीचर/अमेजिंग-फीचर`) -3. अपने परिवर्तन प्रतिबद्ध करें (`गिट कमिट -एम 'अद्भुत सुविधा जोड़ें'`) -4. शाखा में पुश करें (`गिट पुश ओरिजिन फीचर/अद्भुत-फीचर`) -5. एक पुल अनुरोध खोलें +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -विस्तृत दिशानिर्देशों के लिए [CONTRIBUTING.md](CONTRIBUTING.md) देखें।### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1994,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -**[9router](https://github.com/decolua/9router)**को**[decolua](https://github.com/decolua)**द्वारा विशेष धन्यवाद - मूल परियोजना जिसने इस फोर्क को प्रेरित किया। ओमनीरूट अतिरिक्त सुविधाओं, मल्टी-मोडल एपीआई और पूर्ण टाइपस्क्रिप्ट पुनर्लेखन के साथ उस अविश्वसनीय नींव पर आधारित है। +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -**[CLIPrxyAPI](https://github.com/router-for-me/CLIPrxyAPI)**को विशेष धन्यवाद - मूल गो कार्यान्वयन जिसने इस जावास्क्रिप्ट पोर्ट को प्रेरित किया।--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## लाइसेंस -एमआईटी लाइसेंस - विवरण के लिए [लाइसेंस](लाइसेंस) देखें।--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/in/docs/ARCHITECTURE.md b/docs/i18n/in/docs/ARCHITECTURE.md index a743ee1a54..b98e57fd7f 100644 --- a/docs/i18n/in/docs/ARCHITECTURE.md +++ b/docs/i18n/in/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_अंतिम अद्यतन: 2026-03-28_## Executive Summary -ओमनीरूट एक स्थानीय एआई रूटिंग गेटवे और नेक्स्ट.जेएस पर निर्मित डैशबोर्ड है। -यह एक एकल ओपनएआई-संगत एंडपॉइंट (`/v1/*`) प्रदान करता है और अनुवाद, फ़ॉलबैक, टोकन रिफ्रेश और उपयोग ट्रैकिंग के साथ कई अपस्ट्रीम प्रदाताओं के बीच ट्रैफ़िक को रूट करता है। -मुख्य क्षमताएं: +_Last updated: 2026-03-28_ -- सीएलआई/टूल्स के लिए ओपनएआई-संगत एपीआई सतह (28 प्रदाता) -- प्रदाता प्रारूपों में अनुरोध/प्रतिक्रिया अनुवाद -- मॉडल कॉम्बो फ़ॉलबैक (मल्टी-मॉडल अनुक्रम) -- खाता-स्तरीय फ़ॉलबैक (प्रति प्रदाता बहु-खाता) -- OAuth + एपीआई-कुंजी प्रदाता कनेक्शन प्रबंधन -- `/v1/embeddings` के माध्यम से एम्बेडिंग पीढ़ी (6 प्रदाता, 9 मॉडल) -- `/v1/images/पीढ़ी` के माध्यम से छवि निर्माण (4 प्रदाता, 9 मॉडल) -- तर्क मॉडल के लिए थिंक टैग पार्सिंग (`<थिंक>...`) सोचें -- सख्त ओपनएआई एसडीके संगतता के लिए प्रतिक्रिया स्वच्छता -- क्रॉस-प्रदाता अनुकूलता के लिए भूमिका सामान्यीकरण (डेवलपर→सिस्टम, सिस्टम→उपयोगकर्ता)। -- संरचित आउटपुट रूपांतरण (json_schema → जेमिनी रिस्पॉन्सस्कीमा) -- प्रदाताओं, चाबियाँ, उपनाम, कॉम्बो, सेटिंग्स, मूल्य निर्धारण के लिए स्थानीय दृढ़ता -- उपयोग/लागत ट्रैकिंग और अनुरोध लॉगिंग -- मल्टी-डिवाइस/स्टेट सिंक के लिए वैकल्पिक क्लाउड सिंक -- एपीआई एक्सेस नियंत्रण के लिए आईपी अनुमति सूची/ब्लॉकलिस्ट -- सोच बजट प्रबंधन (पासथ्रू/ऑटो/कस्टम/अनुकूली) -- वैश्विक प्रणाली शीघ्र इंजेक्शन -- सत्र ट्रैकिंग और फ़िंगरप्रिंटिंग -- प्रदाता-विशिष्ट प्रोफाइल के साथ प्रति-खाता बढ़ी हुई दर सीमित करना -- प्रदाता लचीलेपन के लिए सर्किट ब्रेकर पैटर्न -- म्यूटेक्स लॉकिंग के साथ एंटी-थंडरिंग झुंड सुरक्षा -- हस्ताक्षर-आधारित अनुरोध डिडुप्लीकेशन कैश -- डोमेन परत: मॉडल उपलब्धता, लागत नियम, फ़ॉलबैक नीति, लॉकआउट नीति -- डोमेन स्थिति दृढ़ता (फ़ॉलबैक, बजट, लॉकआउट, सर्किट ब्रेकर के लिए SQLite राइट-थ्रू कैश) -- केंद्रीकृत अनुरोध मूल्यांकन के लिए नीति इंजन (लॉकआउट → बजट → फ़ॉलबैक) -- p50/p95/p99 विलंबता एकत्रीकरण के साथ टेलीमेट्री का अनुरोध करें -- एंड-टू-एंड ट्रेसिंग के लिए सहसंबंध आईडी (एक्स-रिक्वेस्ट-आईडी)। -- एपीआई कुंजी के अनुसार ऑप्ट-आउट के साथ अनुपालन ऑडिट लॉगिंग -- एलएलएम गुणवत्ता आश्वासन के लिए इवल फ्रेमवर्क -- वास्तविक समय सर्किट ब्रेकर स्थिति के साथ लचीलापन यूआई डैशबोर्ड -- मॉड्यूलर OAuth प्रदाता (`src/lib/oauth/providers/` के अंतर्गत 12 व्यक्तिगत मॉड्यूल) +## Executive Summary -प्राथमिक रनटाइम मॉडल: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- `src/app/api/*` के अंतर्गत Next.js ऐप रूट डैशबोर्ड एपीआई और संगतता एपीआई दोनों को लागू करते हैं -- `src/sse/*` + `open-sse/*` में एक साझा SSE/रूटिंग कोर प्रदाता निष्पादन, अनुवाद, स्ट्रीमिंग, फ़ॉलबैक और उपयोग को संभालता है## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- स्थानीय गेटवे रनटाइम -- डैशबोर्ड प्रबंधन एपीआई -- प्रदाता प्रमाणीकरण और टोकन ताज़ा करें -- अनुवाद और एसएसई स्ट्रीमिंग का अनुरोध करें -- स्थानीय स्थिति + उपयोग की दृढ़ता -- वैकल्पिक क्लाउड सिंक ऑर्केस्ट्रेशन### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- `NEXT_PUBLIC_CLOUD_URL` के पीछे क्लाउड सेवा कार्यान्वयन -- स्थानीय प्रक्रिया के बाहर प्रदाता एसएलए/नियंत्रण विमान -- बाहरी सीएलआई बायनेरिज़ स्वयं (क्लाउड सीएलआई, कोडेक्स सीएलआई, आदि)## Dashboard Surface (Current) +### Out of Scope -`src/app/(डैशबोर्ड)/डैशबोर्ड/` के अंतर्गत मुख्य पृष्ठ: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/डैशबोर्ड` - त्वरित शुरुआत + प्रदाता अवलोकन -- `/डैशबोर्ड/एंडपॉइंट` - एंडपॉइंट प्रॉक्सी + एमसीपी + ए2ए + एपीआई एंडपॉइंट टैब -- `/डैशबोर्ड/प्रदाता` - प्रदाता कनेक्शन और क्रेडेंशियल -- `/डैशबोर्ड/कॉम्बोस` - कॉम्बो रणनीतियाँ, टेम्पलेट, मॉडल रूटिंग नियम -- `/डैशबोर्ड/लागत` - लागत एकत्रीकरण और मूल्य निर्धारण दृश्यता -- `/डैशबोर्ड/एनालिटिक्स` - उपयोग विश्लेषण और मूल्यांकन -- `/डैशबोर्ड/सीमाएँ` - कोटा/दर नियंत्रण -- `/डैशबोर्ड/क्ली-टूल्स` - सीएलआई ऑनबोर्डिंग, रनटाइम डिटेक्शन, कॉन्फिग जेनरेशन -- `/डैशबोर्ड/एजेंट` - पता चला एसीपी एजेंट + कस्टम एजेंट पंजीकरण -- `/डैशबोर्ड/मीडिया` - छवि/वीडियो/संगीत खेल का मैदान -- `/डैशबोर्ड/सर्च-टूल्स` - खोज प्रदाता परीक्षण और इतिहास -- `/डैशबोर्ड/स्वास्थ्य` - अपटाइम, सर्किट ब्रेकर, दर सीमा -- `/डैशबोर्ड/लॉग्स` - अनुरोध/प्रॉक्सी/ऑडिट/कंसोल लॉग -- `/डैशबोर्ड/सेटिंग्स` - सिस्टम सेटिंग्स टैब (सामान्य, रूटिंग, कॉम्बो डिफ़ॉल्ट, आदि) -- `/डैशबोर्ड/एपीआई-मैनेजर` - एपीआई कुंजी जीवनचक्र और मॉडल अनुमतियाँ## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -मुख्य निर्देशिकाएँ: +Main directories: -- अनुकूलता एपीआई के लिए `src/app/api/v1/*` और `src/app/api/v1beta/*` -- प्रबंधन/कॉन्फ़िगरेशन एपीआई के लिए `src/app/api/*` -- अगला `next.config.mjs` मैप `/v1/*` से `/api/v1/*` में फिर से लिखता है +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -महत्वपूर्ण अनुकूलता मार्ग: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` - इसमें `कस्टम: ट्रू` के साथ कस्टम मॉडल शामिल हैं -- `src/app/api/v1/embeddings/route.ts` - एम्बेडिंग जेनरेशन (6 प्रदाता) -- `src/app/api/v1/images/जेनरेशन/रूट.ts` - छवि निर्माण (एंटीग्रेविटी/नेबियस सहित 4+ प्रदाता) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` - प्रति-प्रदाता समर्पित चैट -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` - प्रति-प्रदाता समर्पित एम्बेडिंग -- `src/app/api/v1/providers/[provider]/images/nations/route.ts` - प्रति-प्रदाता समर्पित छवियां +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` - `src/app/api/v1beta/models/[...path]/route.ts` -प्रबंधन डोमेन: +Management domains: -- प्रामाणिक/सेटिंग्स: `src/app/api/auth/*`, `src/app/api/settings/*` -- प्रदाता/कनेक्शन: `src/app/api/प्रदाता*` -- प्रदाता नोड्स: `src/app/api/provider-nodes*` -- कस्टम मॉडल: `src/app/api/provider-models` (प्राप्त करें/पोस्ट करें/हटाएं) -- मॉडल कैटलॉग: `src/app/api/models/route.ts` (GET) -- प्रॉक्सी कॉन्फ़िगरेशन: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- कुंजी/उपनाम/कॉम्बोस/मूल्य निर्धारण: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- उपयोग: `src/app/api/usage/*` -- सिंक/क्लाउड: `src/app/api/sync/*`, `src/app/api/cloud/*` -- सीएलआई टूलींग सहायक: `src/app/api/cli-tools/*` -- आईपी फ़िल्टर: `src/app/api/settings/ip-filter` (प्राप्त/पुट) -- सोच बजट: `src/app/api/settings/thinking-budget` (प्राप्त/पुट) -- सिस्टम प्रॉम्प्ट: `src/app/api/settings/system-prompt` (GET/PUT) -- सत्र: `src/app/api/sessions` (प्राप्त करें) -- दर सीमा: `src/app/api/दर-सीमा` (प्राप्त करें) -- लचीलापन: `src/app/api/resilience` (GET/PATCH) - प्रदाता प्रोफाइल, सर्किट ब्रेकर, दर सीमा स्थिति -- लचीलापन रीसेट: `src/app/api/resilience/reset` (POST) - रीसेट ब्रेकर + कूलडाउन -- कैश आँकड़े: `src/app/api/cache/stats` (प्राप्त करें/हटाएँ) -- मॉडल उपलब्धता: `src/app/api/मॉडल/उपलब्धता` (प्राप्त करें/पोस्ट करें) -- टेलीमेट्री: `src/app/api/टेलीमेट्री/सारांश` (प्राप्त करें) -- बजट: `src/app/api/usage/budget` (प्राप्त करें/पोस्ट करें) -- फ़ॉलबैक चेन: `src/app/api/फ़ॉलबैक/चेन` (प्राप्त करें/पोस्ट करें/हटाएं) -- अनुपालन ऑडिट: `src/app/api/compliance/audit-log` (GET) -- इवल्स: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- नीतियां: `src/app/api/policies` (प्राप्त करें/पोस्ट करें)## 2) SSE + Translation Core +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -मुख्य प्रवाह मॉड्यूल: +## 2) SSE + Translation Core -- प्रविष्टि: `src/sse/handlers/chat.ts` -- कोर ऑर्केस्ट्रेशन: `open-sse/handlers/chatCore.ts` -- प्रदाता निष्पादन एडाप्टर: `ओपन-एसएसई/निष्पादक/*` -- प्रारूप पहचान/प्रदाता कॉन्फिगरेशन: `open-sse/services/provider.ts` -- मॉडल पार्स/रिज़ॉल्यूशन: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- खाता फ़ॉलबैक तर्क: `open-sse/services/accountFallback.ts` -- अनुवाद रजिस्ट्री: `open-sse/translator/index.ts` -- स्ट्रीम परिवर्तन: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- उपयोग निष्कर्षण/सामान्यीकरण: `open-sse/utils/usageTracking.ts` -- टैग पार्सर सोचें: `open-sse/utils/thinkTagParser.ts` -- एंबेडिंग हैंडलर: `open-sse/handlers/embeddings.ts` -- एंबेडिंग प्रदाता रजिस्ट्री: `open-sse/config/embeddingRegistry.ts` -- छवि निर्माण हैंडलर: `open-sse/handlers/imageGeneration.ts` -- छवि प्रदाता रजिस्ट्री: `open-sse/config/imageRegistry.ts` -- रिस्पॉन्स सेनिटाइजेशन: `ओपन-एसएसई/हैंडलर्स/रिस्पांससैनिटाइजर.टीएस` -- भूमिका सामान्यीकरण: `open-sse/services/roleNormalizer.ts` +Main flow modules: -सेवाएँ (व्यावसायिक तर्क): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- खाता चयन/स्कोरिंग: `open-sse/services/accountSelector.ts` -- संदर्भ जीवनचक्र प्रबंधन: `open-sse/services/contextManager.ts` -- आईपी फ़िल्टर प्रवर्तन: `open-sse/services/ipFilter.ts` -- सत्र ट्रैकिंग: `open-sse/services/sessionManager.ts` -- डुप्लिकेशन अनुरोध: `open-sse/services/signatureCache.ts` -- सिस्टम प्रॉम्प्ट इंजेक्शन: `open-sse/services/systemPrompt.ts` -- सोच बजट प्रबंधन: `open-sse/services/thinkingBudget.ts` -- वाइल्डकार्ड मॉडल रूटिंग: `open-sse/services/wildcardRouter.ts` -- दर सीमा प्रबंधन: `open-sse/services/rateLimitManager.ts` -- सर्किट ब्रेकर: `open-sse/services/circuitBreaker.ts` +Services (business logic): -डोमेन परत मॉड्यूल: +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- मॉडल उपलब्धता: `src/lib/domain/modelAvailability.ts` -- लागत नियम/बजट: `src/lib/domain/costRules.ts` -- फ़ॉलबैक नीति: `src/lib/domain/fallbackPolicy.ts` -- कॉम्बो रिज़ॉल्वर: `src/lib/domain/comboResolver.ts` -- लॉकआउट नीति: `src/lib/domain/lockoutPolicy.ts` -- नीति इंजन: `src/domain/policyEngine.ts` - केंद्रीकृत लॉकआउट → बजट → फ़ॉलबैक मूल्यांकन -- त्रुटि कोड कैटलॉग: `src/lib/domain/errorCodes.ts` -- अनुरोध आईडी: `src/lib/domain/requestId.ts` -- फ़ेच टाइमआउट: `src/lib/domain/fetchTimeout.ts` -- अनुरोध टेलीमेट्री: `src/lib/domain/requestTelemetry.ts` -- अनुपालन/ऑडिट: `src/lib/domain/compliance/index.ts` -- इवल रनर: `src/lib/domain/evalRunner.ts` -- डोमेन स्थिति दृढ़ता: `src/lib/db/domainState.ts` - फ़ॉलबैक चेन, बजट, लागत इतिहास, लॉकआउट स्थिति, सर्किट ब्रेकर के लिए SQLite CRUD +Domain layer modules: -OAuth प्रदाता मॉड्यूल (`src/lib/oauth/providers/` के अंतर्गत 12 व्यक्तिगत फ़ाइलें): +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -- रजिस्ट्री सूचकांक: `src/lib/oauth/providers/index.ts` -- व्यक्तिगत प्रदाता: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` -- पतला आवरण: `src/lib/oauth/providers.ts` - अलग-अलग मॉड्यूल से पुनः निर्यात## 3) Persistence Layer +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -प्राथमिक अवस्था DB (SQLite): +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -- कोर इन्फ्रा: `src/lib/db/core.ts` (बेहतर-sqlite3, माइग्रेशन, वाल) -- पुनः निर्यात पहलू: `src/lib/localDb.ts` (कॉलर्स के लिए पतली अनुकूलता परत) -- फ़ाइल: `${DATA_DIR}/storage.sqlite` (या `$XDG_CONFIG_HOME/omniroute/storage.sqlite` सेट होने पर, अन्यथा `~/.omniroute/storage.sqlite`) -- इकाइयां (टेबल + केवी नेमस्पेस): प्रोवाइडरकनेक्शन्स, प्रोवाइडरनोड्स, मॉडलएलियासेस, कॉम्बो, एपीआईकीज़, सेटिंग्स, मूल्य निर्धारण,**कस्टममॉडल**,**प्रॉक्सीकॉन्फिग**,**आईपीफिल्टर**,**थिंकिंगबजट**,**सिस्टमप्रॉम्प्ट** +## 3) Persistence Layer -उपयोग दृढ़ता: +Primary state DB (SQLite): -- मुखौटा: `src/lib/usageDb.ts` (`src/lib/usage/*` में विघटित मॉड्यूल) -- `storage.sqlite` में SQLite तालिकाएँ: `usage_history`, `call_logs`, `proxy_logs` -- अनुकूलता/डीबग के लिए वैकल्पिक फ़ाइल कलाकृतियाँ बनी रहती हैं (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- मौजूद होने पर लीगेसी JSON फ़ाइलें स्टार्टअप माइग्रेशन द्वारा SQLite में माइग्रेट की जाती हैं +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -डोमेन स्थिति DB (SQLite): +Usage persistence: -- `src/lib/db/domainState.ts` - डोमेन स्थिति के लिए CRUD संचालन -- टेबल्स (`src/lib/db/core.ts` में निर्मित): `domain_fallback_चेन्स`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- राइट-थ्रू कैश पैटर्न: इन-मेमोरी मैप्स रनटाइम पर आधिकारिक होते हैं; उत्परिवर्तन SQLite के साथ समकालिक रूप से लिखे जाते हैं; कोल्ड स्टार्ट पर राज्य को डीबी से बहाल किया जाता है## 4) Auth + Security Surfaces +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- डैशबोर्ड कुकी प्रमाणीकरण: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- एपीआई कुंजी निर्माण/सत्यापन: `src/shared/utils/apiKey.ts` -- प्रदाता रहस्य `providerConnections` प्रविष्टियों में बने रहे -- `open-sse/utils/proxyFetch.ts` (env vars) और `open-sse/utils/networkProxy.ts` (प्रति-प्रदाता या वैश्विक रूप से कॉन्फ़िगर करने योग्य) के माध्यम से आउटबाउंड प्रॉक्सी समर्थन## 5) Cloud Sync +Domain State DB (SQLite): -- शेड्यूलर init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- आवधिक कार्य: `src/shared/services/cloudSyncScheduler.ts` -- आवधिक कार्य: `src/shared/services/modelSyncScheduler.ts` -- नियंत्रण मार्ग: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -फ़ॉलबैक निर्णय स्थिति कोड और त्रुटि-संदेश अनुमानों का उपयोग करके `open-sse/services/accountFallback.ts` द्वारा संचालित होते हैं। कॉम्बो रूटिंग एक अतिरिक्त गार्ड जोड़ता है: प्रदाता-स्कोप्ड 400s जैसे अपस्ट्रीम सामग्री-ब्लॉक और भूमिका-सत्यापन विफलताओं को मॉडल-स्थानीय विफलताओं के रूप में माना जाता है ताकि बाद में कॉम्बो लक्ष्य अभी भी चल सकें।## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -लाइव ट्रैफ़िक के दौरान रिफ्रेश को निष्पादक `refreshCredentials()` के माध्यम से `open-sse/handlers/chatCore.ts` के अंदर निष्पादित किया जाता है।## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -क्लाउड सक्षम होने पर आवधिक सिंक `CloudSyncScheduler` द्वारा ट्रिगर किया जाता है।## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -भौतिक भंडारण फ़ाइलें: +Physical storage files: -- प्राथमिक रनटाइम DB: `${DATA_DIR}/storage.sqlite` -- अनुरोध लॉग लाइनें: `${DATA_DIR}/log.txt` (कॉम्पैट/डीबग आर्टिफैक्ट) -- संरचित कॉल पेलोड अभिलेखागार: `${DATA_DIR}/call_logs/` -- वैकल्पिक अनुवादक/अनुरोध डिबग सत्र: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: अनुकूलता एपीआई -- `src/app/api/v1/providers/[provider]/*`: प्रति-प्रदाता समर्पित मार्ग (चैट, एम्बेडिंग, चित्र) -- `src/app/api/providers*`: प्रदाता CRUD, सत्यापन, परीक्षण -- `src/app/api/provider-nodes*`: कस्टम संगत नोड प्रबंधन -- `src/app/api/provider-models`: कस्टम मॉडल प्रबंधन (CRUD) -- `src/app/api/models/route.ts`: मॉडल कैटलॉग एपीआई (उपनाम + कस्टम मॉडल) -- `src/app/api/oauth/*`: OAuth/डिवाइस-कोड प्रवाह -- `src/app/api/keys*`: स्थानीय एपीआई कुंजी जीवनचक्र -- `src/app/api/models/alias`: उपनाम प्रबंधन -- `src/app/api/combos*`: फ़ॉलबैक कॉम्बो प्रबंधन -- `src/app/api/pricing`: लागत गणना के लिए मूल्य निर्धारण ओवरराइड हो जाता है -- `src/app/api/settings/proxy`: प्रॉक्सी कॉन्फ़िगरेशन (प्राप्त/पुट/हटाएं) -- `src/app/api/settings/proxy/test`: आउटबाउंड प्रॉक्सी कनेक्टिविटी टेस्ट (POST) -- `src/app/api/usage/*`: एपीआई का उपयोग और लॉग -- `src/app/api/sync/*` + `src/app/api/cloud/*`: क्लाउड सिंक और क्लाउड-फेसिंग हेल्पर्स -- `src/app/api/cli-tools/*`: स्थानीय सीएलआई कॉन्फ़िगरेशन लेखक/चेकर्स -- `src/app/api/settings/ip-filter`: आईपी अनुमति सूची/ब्लॉकलिस्ट (प्राप्त/पुट) -- `src/app/api/settings/thinking-budget`: थिंकिंग टोकन बजट कॉन्फ़िगरेशन (GET/PUT) -- `src/app/api/settings/system-prompt`: ग्लोबल सिस्टम प्रॉम्प्ट (GET/PUT) -- `src/app/api/sessions`: सक्रिय सत्र सूची (GET) -- `src/app/api/rate-limits`: प्रति-खाता दर सीमा स्थिति (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: अनुरोध पार्स, कॉम्बो हैंडलिंग, खाता चयन लूप -- `ओपन-एसएसई/हैंडलर/चैटकोर.टीएस`: अनुवाद, निष्पादक प्रेषण, पुनः प्रयास/रीफ्रेश हैंडलिंग, स्ट्रीम सेटअप -- `ओपन-एसएसई/निष्पादक/*`: प्रदाता-विशिष्ट नेटवर्क और प्रारूप व्यवहार### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: अनुवादक रजिस्ट्री और ऑर्केस्ट्रेशन -- अनुवादकों से अनुरोध: `ओपन-एसएसई/अनुवादक/अनुरोध/*` -- प्रतिक्रिया अनुवादक: `ओपन-एसएसई/अनुवादक/प्रतिक्रिया/*` -- प्रारूप स्थिरांक: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: SQLite पर लगातार कॉन्फ़िगरेशन/स्थिति और डोमेन दृढ़ता -- `src/lib/localDb.ts`: डीबी मॉड्यूल के लिए अनुकूलता पुनः निर्यात -- `src/lib/usageDb.ts`: SQLite तालिकाओं के शीर्ष पर उपयोग इतिहास/कॉल लॉग मुखौटा## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -प्रत्येक प्रदाता के पास `BaseExecutor` (`open-sse/executors/base.ts` में) का विस्तार करने वाला एक विशेष निष्पादक होता है, जो URL निर्माण, हेडर निर्माण, घातीय बैकऑफ़ के साथ पुनः प्रयास, क्रेडेंशियल रिफ्रेश हुक और `execute()` ऑर्केस्ट्रेशन विधि प्रदान करता है। +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| निष्पादक | प्रदाता(ओं) | विशेष हैंडलिंग | -| --------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------- | -| `डिफ़ॉल्ट निष्पादक` | ओपनएआई, क्लाउड, जेमिनी, क्वेन, क्यूडर, ओपनराउटर, जीएलएम, किमी, मिनीमैक्स, डीपसीक, ग्रोक, एक्सएआई, मिस्ट्रल, पर्प्लेक्सिटी, टुगेदर, फायरवर्क्स, सेरेब्रा, कोहेरे, एनवीआईडीआईए | प्रति प्रदाता डायनामिक यूआरएल/हेडर कॉन्फिगरेशन | -| 'एंटीग्रेविटी एक्ज़ीक्यूटर' | गूगल एंटीग्रेविटी | कस्टम प्रोजेक्ट/सत्र आईडी, पुनः प्रयास करें-पार्सिंग के बाद | -| `कोडेक्स एक्ज़ीक्यूटर` ​​ | ओपनएआई कोडेक्स | सिस्टम निर्देश इंजेक्ट करता है, तर्क करने का प्रयास करता है | -| `कर्सर निष्पादक` | कर्सर आईडीई | कनेक्टआरपीसी प्रोटोकॉल, प्रोटोबफ एन्कोडिंग, चेकसम के माध्यम से हस्ताक्षर करने का अनुरोध | -| 'GithubExecutor' | गिटहब कोपायलट | कोपायलट टोकन ताज़ा करें, VSCode-नकल हेडर | -| `कीरो एक्ज़ीक्यूटर` ​​ | एडब्ल्यूएस कोडव्हिस्परर/किरो | एडब्ल्यूएस इवेंटस्ट्रीम बाइनरी प्रारूप → एसएसई रूपांतरण | -| `जेमिनीसीएलआईएक्सक्यूटर` ​​ | जेमिनी सीएलआई | Google OAuth टोकन ताज़ा चक्र | +### Persistence -अन्य सभी प्रदाता (कस्टम संगत नोड्स सहित) `DefaultExecutor` का उपयोग करते हैं।## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| प्रदाता | प्रारूप | प्रामाणिक | स्ट्रीम | नॉन-स्ट्रीम | टोकन ताज़ा करें | उपयोग एपीआई | -| --------------------- | -------------------- | ------------------------ | ----------------- | ----------- | --------------- | ------------------- | ------------------------------ | -| क्लाउड | क्लाउड | एपीआई कुंजी / OAuth | ✅ | ✅ | ✅ | ⚠️ केवल एडमिन | -| मिथुन | मिथुन | एपीआई कुंजी / OAuth | ✅ | ✅ | ✅ | ⚠️ क्लाउड कंसोल | -| जेमिनी सीएलआई | मिथुन-क्ली | OAuth | ✅ | ✅ | ✅ | ⚠️ क्लाउड कंसोल | -| प्रतिगुरुत्वाकर्षण | प्रतिगुरुत्वाकर्षण | OAuth | ✅ | ✅ | ✅ | ✅ पूर्ण कोटा एपीआई | -| ओपनएआई | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| कोडेक्स | openai-प्रतिक्रियाएं | OAuth | ✅ मजबूर | ❌ | ✅ | ✅ दर सीमा | -| गिटहब कोपायलट | ओपनाई | OAuth + सहपायलट टोकन | ✅ | ✅ | ✅ | ✅ कोटा स्नैपशॉट | -| कर्सर | कर्सर | कस्टम चेकसम | ✅ | ✅ | ❌ | ❌ | -| किरो | किरो | एडब्ल्यूएस एसएसओ ओआईडीसी | ✅ (इवेंटस्ट्रीम) | ❌ | ✅ | ✅ उपयोग सीमा | -| क्वेन | ओपनाई | OAuth | ✅ | ✅ | ✅ | ⚠️ प्रति अनुरोध | -| कोडर | ओपनाई | OAuth (बेसिक) | ✅ | ✅ | ✅ | ⚠️ प्रति अनुरोध | -| ओपनराउटर | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| जीएलएम/किमी/मिनीमैक्स | क्लाउड | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| डीपसीक | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| ग्रोक | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| एक्सएआई (ग्रोक) | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| मिस्ट्रल | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| उलझन | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| एक साथ एआई | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| आतिशबाजी एआई | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| सेरेब्रस | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| सहभागी | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | -| एनवीडिया एनआईएम | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -पता लगाए गए स्रोत प्रारूपों में शामिल हैं: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. -- `ओपनाई` -- `ओपनई-प्रतिक्रियाएँ` -- `क्लाउड` -- 'मिथुन' +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | -लक्ष्य प्रारूपों में शामिल हैं: +All other providers (including custom compatible nodes) use the `DefaultExecutor`. -- ओपनएआई चैट/प्रतिक्रियाएं - -क्लाउड -- मिथुन/मिथुन-सीएलआई/एंटीग्रेविटी लिफाफा -- किरो -- कर्सर +## Provider Compatibility Matrix -अनुवाद**हब प्रारूप के रूप में ओपनएआई**का उपयोग करते हैं - सभी रूपांतरण मध्यवर्ती के रूप में ओपनएआई से गुजरते हैं:``` +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: + +- `openai` +- `openai-responses` +- `claude` +- `gemini` + +Target formats include: + +- OpenAI chat/Responses +- Claude +- Gemini/Gemini-CLI/Antigravity envelope +- Kiro +- Cursor + +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -स्रोत पेलोड आकार और प्रदाता लक्ष्य प्रारूप के आधार पर अनुवादों का चयन गतिशील रूप से किया जाता है। +Additional processing layers in the translation pipeline: -अनुवाद पाइपलाइन में अतिरिक्त प्रसंस्करण परतें: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**प्रतिक्रिया स्वच्छता**- सख्त एसडीके अनुपालन सुनिश्चित करने के लिए ओपनएआई-प्रारूप प्रतिक्रियाओं (स्ट्रीमिंग और गैर-स्ट्रीमिंग दोनों) से गैर-मानक फ़ील्ड हटा देता है --**भूमिका सामान्यीकरण**- गैर-ओपनएआई लक्ष्यों के लिए `डेवलपर` → `सिस्टम` को रूपांतरित करता है; सिस्टम भूमिका को अस्वीकार करने वाले मॉडलों के लिए `सिस्टम` → `उपयोगकर्ता` को मर्ज करता है (जीएलएम, ईआरएनआईई) --**टैग निष्कर्षण के बारे में सोचें**- पार्स `<सोच>...` सामग्री से `reasoning_content` फ़ील्ड में ब्लॉक करता है --**संरचित आउटपुट**- OpenAI `response_format.json_schema` को जेमिनी के `responseMimeType` + `responseSchema` में परिवर्तित करता है## Supported API Endpoints +## Supported API Endpoints -| समापन बिंदु | प्रारूप | हैंडलर | -| -------------------------------------------------- | ------------------ | ---------------------------------------------------------------------------------- | -| `पोस्ट /v1/चैट/समापन` | ओपनएआई चैट | `src/sse/handlers/chat.ts` | -| `पोस्ट /v1/संदेश` | क्लाउड संदेश | वही हैंडलर (स्वतः पता चला) | -| `पोस्ट /v1/प्रतिक्रियाएँ` | ओपनएआई प्रतिक्रियाएँ | `open-sse/handlers/responsesHandler.ts` | -| `पोस्ट /v1/एम्बेडिंग` | ओपनएआई एंबेडिंग्स | `open-sse/handlers/embeddings.ts` | -| `प्राप्त करें /v1/एम्बेडिंग्स` | मॉडल सूची | एपीआई मार्ग | -| `पोस्ट /v1/छवियां/पीढ़ी` | OpenAI छवियाँ | `ओपन-एसएसई/हैंडलर/इमेजजेनरेशन.टीएस` | -| `प्राप्त करें /v1/छवियां/पीढ़ी` | मॉडल सूची | एपीआई मार्ग | -| `पोस्ट /v1/प्रदाता/{प्रदाता}/चैट/समापन` | ओपनएआई चैट | मॉडल सत्यापन के साथ प्रति-प्रदाता समर्पित | -| `पोस्ट /v1/प्रदाता/{प्रदाता}/एम्बेडिंग्स` | ओपनएआई एंबेडिंग्स | मॉडल सत्यापन के साथ प्रति-प्रदाता समर्पित | -| `पोस्ट /v1/प्रदाता/{प्रदाता}/छवियां/पीढ़ी` | OpenAI छवियाँ | मॉडल सत्यापन के साथ प्रति-प्रदाता समर्पित | -| `POST /v1/messages/count_tokens` | क्लाउड टोकन गिनती | एपीआई मार्ग | -| `प्राप्त करें /v1/मॉडल` | OpenAI मॉडल सूची | एपीआई मार्ग (चैट + एम्बेडिंग + छवि + कस्टम मॉडल) | -| `प्राप्त करें /एपीआई/मॉडल/कैटलॉग` | कैटलॉग | प्रदाता + प्रकार | द्वारा समूहीकृत सभी मॉडल -| `POST /v1beta/models/*:streamGenerateContent` | मिथुन राशि के जातक | एपीआई मार्ग | -| `प्राप्त/पुट/डिलीट /एपीआई/सेटिंग्स/प्रॉक्सी` | प्रॉक्सी कॉन्फिग | नेटवर्क प्रॉक्सी कॉन्फ़िगरेशन | -| `पोस्ट /एपीआई/सेटिंग्स/प्रॉक्सी/टेस्ट` | प्रॉक्सी कनेक्टिविटी | प्रॉक्सी स्वास्थ्य/कनेक्टिविटी परीक्षण समापन बिंदु | -| `प्राप्त करें/पोस्ट करें/हटाएं /एपीआई/प्रदाता-मॉडल` | प्रदाता मॉडल | प्रदाता मॉडल मेटाडेटा समर्थन कस्टम और प्रबंधित उपलब्ध मॉडल |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -बाईपास हैंडलर (`ओपन-एसएसई/यूटिल्स/बायपासहैंडलर.टीएस`) क्लाउड सीएलआई से ज्ञात "थ्रोअवे" अनुरोधों को रोकता है - वार्मअप पिंग, शीर्षक निष्कर्षण और टोकन गिनती - और अपस्ट्रीम प्रदाता टोकन का उपभोग किए बिना एक**नकली प्रतिक्रिया**देता है। यह तभी ट्रिगर होता है जब `User-Agent` में `claude-cli` होता है।## Request Logger Pipeline +## Bypass Handler -अनुरोध लकड़हारा (`open-sse/utils/requestLogger.ts`) एक 7-चरण डीबग लॉगिंग पाइपलाइन प्रदान करता है, जो डिफ़ॉल्ट रूप से अक्षम है, `ENABLE_REQUEST_LOGS=true` के माध्यम से सक्षम है:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -प्रत्येक अनुरोध सत्र के लिए फ़ाइलें `/logs//` पर लिखी जाती हैं।## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- क्षणिक/दर/प्रामाणिक त्रुटियों पर प्रदाता खाता ठंडा हो गया -- अनुरोध विफल होने से पहले खाता फ़ॉलबैक -- वर्तमान मॉडल/प्रदाता पथ समाप्त होने पर कॉम्बो मॉडल फ़ॉलबैक## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- ताज़ा करने योग्य प्रदाताओं के लिए पुनः प्रयास के साथ पूर्व-जांच और ताज़ा करें -- कोर पथ में ताज़ा प्रयास के बाद 401/403 पुनः प्रयास करें## 3) Stream Safety +## 2) Token Expiry -- डिस्कनेक्ट-अवेयर स्ट्रीम नियंत्रक -- एंड-ऑफ-स्ट्रीम फ्लश और `[DONE]` हैंडलिंग के साथ अनुवाद स्ट्रीम -- प्रदाता उपयोग मेटाडेटा अनुपलब्ध होने पर उपयोग अनुमान फ़ॉलबैक## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- समन्वयन त्रुटियाँ सामने आती हैं लेकिन स्थानीय रनटाइम जारी रहता है -- शेड्यूलर में पुनः प्रयास-सक्षम तर्क है, लेकिन आवधिक निष्पादन वर्तमान में डिफ़ॉल्ट रूप से एकल-प्रयास सिंक को कॉल करता है## 5) Data Integrity +## 3) Stream Safety -- स्टार्टअप पर SQLite स्कीमा माइग्रेशन और ऑटो-अपग्रेड हुक -- लीगेसी JSON → SQLite माइग्रेशन संगतता पथ## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -रनटाइम दृश्यता स्रोत: +## 4) Cloud Sync Degradation -- कंसोल `src/sse/utils/logger.ts` से लॉग करता है -- SQLite में प्रति-अनुरोध उपयोग समुच्चय (`use_history`, `call_logs`, `proxy_logs`) -- जब `settings.detailed_logs_enabled=true` होता है तो SQLite (`request_detail_logs`) में चार चरण वाला विस्तृत पेलोड कैप्चर होता है -- पाठ्य अनुरोध स्थिति लॉग इन `log.txt` (वैकल्पिक/कॉम्पैट) -- `ENABLE_REQUEST_LOGS=true` होने पर `लॉग/` के अंतर्गत वैकल्पिक गहन अनुरोध/अनुवाद लॉग -- यूआई खपत के लिए डैशबोर्ड उपयोग समापन बिंदु (`/api/usage/*`)। +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -विस्तृत अनुरोध पेलोड कैप्चर प्रति रूटेड कॉल को चार JSON पेलोड चरणों तक संग्रहीत करता है: +## 5) Data Integrity -- ग्राहक से प्राप्त कच्चा अनुरोध -- अनुवादित अनुरोध वास्तव में अपस्ट्रीम भेजा गया -- प्रदाता प्रतिक्रिया JSON के रूप में पुनर्निर्मित; स्ट्रीम की गई प्रतिक्रियाओं को अंतिम सारांश और स्ट्रीम मेटाडेटा में संकलित किया जाता है -- ओम्निरूट द्वारा लौटाई गई अंतिम ग्राहक प्रतिक्रिया; स्ट्रीम की गई प्रतिक्रियाएँ उसी संक्षिप्त सारांश रूप में संग्रहीत की जाती हैं## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- JWT सीक्रेट (`JWT_SECRET`) डैशबोर्ड सत्र कुकी सत्यापन/हस्ताक्षर को सुरक्षित करता है -- प्रारंभिक पासवर्ड बूटस्ट्रैप (`INITIAL_PASSWORD`) को प्रथम-रन प्रावधान के लिए स्पष्ट रूप से कॉन्फ़िगर किया जाना चाहिए -- एपीआई कुंजी HMAC सीक्रेट (`API_KEY_SECRET`) उत्पन्न स्थानीय एपीआई कुंजी प्रारूप को सुरक्षित करता है -- प्रदाता रहस्य (एपीआई कुंजी/टोकन) स्थानीय डीबी में बने रहते हैं और उन्हें फ़ाइल सिस्टम स्तर पर संरक्षित किया जाना चाहिए -- क्लाउड सिंक एंडपॉइंट एपीआई कुंजी ऑथ + मशीन आईडी सेमेन्टिक्स पर निर्भर करते हैं## Environment and Runtime Matrix +## Observability and Operational Signals -कोड द्वारा सक्रिय रूप से उपयोग किए जाने वाले पर्यावरण चर: +Runtime visibility sources: -- ऐप/ऑथ: `JWT_SECRET`, `INITIAL_PASSWORD` -- भंडारण: `DATA_DIR` -- संगत नोड व्यवहार: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- वैकल्पिक स्टोरेज बेस ओवरराइड (Linux/macOS जब `DATA_DIR` अनसेट होता है): `XDG_CONFIG_HOME` -- सुरक्षा हैशिंग: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- लॉगिंग: `ENABLE_REQUEST_LOGS` -- सिंक/क्लाउड यूआरएल: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- आउटबाउंड प्रॉक्सी: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` और लोअरकेस वेरिएंट -- SOCKS5 फ़ीचर फ़्लैग: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- प्लेटफ़ॉर्म/रनटाइम सहायक (ऐप-विशिष्ट कॉन्फ़िगरेशन नहीं): `एप्लिकेशन डेटा`, `NODE_ENV`, `पोर्ट`, `होस्टनाम`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` और `localDb` लीगेसी फ़ाइल माइग्रेशन के साथ समान आधार निर्देशिका नीति (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) साझा करते हैं। -2. `/api/v1/route.ts` सिमेंटिक बहाव से बचने के लिए `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) द्वारा उपयोग किए जाने वाले समान एकीकृत कैटलॉग बिल्डर को सौंपता है। -3. अनुरोध लकड़हारा सक्षम होने पर पूर्ण हेडर/बॉडी लिखता है; लॉग निर्देशिका को संवेदनशील मानें। -4. क्लाउड व्यवहार सही `NEXT_PUBLIC_BASE_URL` और क्लाउड एंडपॉइंट रीचैबिलिटी पर निर्भर करता है। -5. `ओपन-एसएसई/` निर्देशिका को `@omniroute/ओपन-एसएसई`**एनपीएम वर्कस्पेस पैकेज**के रूप में प्रकाशित किया गया है। स्रोत कोड इसे `@omniroute/open-sse/...` (Next.js `transpilePackages` द्वारा हल) के माध्यम से आयात करता है। इस दस्तावेज़ में फ़ाइल पथ अभी भी स्थिरता के लिए निर्देशिका नाम `open-sse/` का उपयोग करते हैं। -6. डैशबोर्ड में चार्ट सुलभ, इंटरैक्टिव एनालिटिक्स विज़ुअलाइज़ेशन (मॉडल उपयोग बार चार्ट, सफलता दर के साथ प्रदाता ब्रेकडाउन टेबल) के लिए**रिचार्ट्स**(एसवीजी-आधारित) का उपयोग करते हैं। -7. E2E परीक्षण**Playwright**(`test/e2e/`) का उपयोग करते हैं, `npm run test:e2e` के माध्यम से चलते हैं। यूनिट परीक्षण**नोड.जेएस टेस्ट रनर**(`टेस्ट/यूनिट/`) का उपयोग करते हैं, जो `एनपीएम रन टेस्ट: यूनिट` के माध्यम से चलते हैं। `src/` के अंतर्गत स्रोत कोड**टाइपस्क्रिप्ट**(`.ts`/`.tsx`) है; `ओपन-एसएसई/` कार्यक्षेत्र जावास्क्रिप्ट (`.जेएस`) बना हुआ है। -8. सेटिंग्स पृष्ठ को 5 टैब में व्यवस्थित किया गया है: सुरक्षा, रूटिंग (6 वैश्विक रणनीतियाँ: भरण-प्रथम, राउंड-रॉबिन, पी2सी, यादृच्छिक, कम से कम उपयोग किया गया, लागत-अनुकूलित), लचीलापन (संपादन योग्य दर सीमा, सर्किट ब्रेकर, नीतियां), एआई (सोच बजट, सिस्टम प्रॉम्प्ट, प्रॉम्प्ट कैश), उन्नत (प्रॉक्सी)।## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- स्रोत से निर्माण: `एनपीएम रन बिल्ड` -- डॉकर छवि बनाएँ: `docker build -t omniroute।` -- सेवा प्रारंभ करें और सत्यापित करें: -- `प्राप्त करें /एपीआई/सेटिंग्स` -- `प्राप्त करें /api/v1/मॉडल` -- जब `PORT=20128` हो तो CLI लक्ष्य आधार URL `http://:20128/v1` होना चाहिए +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` +- `GET /api/v1/models` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/in/docs/FEATURES.md b/docs/i18n/in/docs/FEATURES.md index b068a0ddf1..a4ab23d256 100644 --- a/docs/i18n/in/docs/FEATURES.md +++ b/docs/i18n/in/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -ओमनीरूट डैशबोर्ड के प्रत्येक अनुभाग के लिए विज़ुअल गाइड।--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -एआई प्रदाता कनेक्शन प्रबंधित करें: OAuth प्रदाता (क्लाउड कोड, कोडेक्स, जेमिनी सीएलआई), एपीआई कुंजी प्रदाता (ग्रोक, डीपसीक, ओपनराउटर), और मुफ्त प्रदाता (क्यूडर, क्वेन, किरो)। किरो खातों में क्रेडिट बैलेंस ट्रैकिंग - शेष क्रेडिट, कुल भत्ता और डैशबोर्ड → उपयोग में दिखाई देने वाली नवीनीकरण तिथि शामिल है।![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -6 रणनीतियों के साथ मॉडल रूटिंग कॉम्बो बनाएं: प्राथमिकता, भारित, राउंड-रॉबिन, यादृच्छिक, कम से कम उपयोग किया गया और लागत-अनुकूलित। प्रत्येक कॉम्बो स्वचालित फ़ॉलबैक के साथ कई मॉडलों को श्रृंखलाबद्ध करता है और इसमें त्वरित टेम्पलेट और तत्परता जांच शामिल होती है।![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -टोकन खपत, लागत अनुमान, गतिविधि हीटमैप, साप्ताहिक वितरण चार्ट और प्रति-प्रदाता विश्लेषण के साथ व्यापक उपयोग विश्लेषण।![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -वास्तविक समय की निगरानी: अपटाइम, मेमोरी, संस्करण, विलंबता प्रतिशत (p50/p95/p99), कैश आँकड़े, और प्रदाता सर्किट ब्रेकर स्थिति।![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -एपीआई अनुवादों को डीबग करने के लिए चार मोड:**प्लेग्राउंड**(फॉर्मेट कनवर्टर),**चैट टेस्टर**(लाइव अनुरोध),**टेस्ट बेंच**(बैच टेस्ट), और**लाइव मॉनिटर**(रियल-टाइम स्ट्रीम)।![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -किसी भी मॉडल का सीधे डैशबोर्ड से परीक्षण करें। प्रदाता, मॉडल और समापन बिंदु का चयन करें, मोनाको संपादक के साथ संकेत लिखें, वास्तविक समय में प्रतिक्रियाओं को स्ट्रीम करें, मध्य-धारा को निरस्त करें, और समय मेट्रिक्स देखें।--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -संपूर्ण डैशबोर्ड के लिए अनुकूलन योग्य रंग थीम। 7 पूर्व निर्धारित रंगों (कोरल, नीला, लाल, हरा, बैंगनी, नारंगी, सियान) में से चुनें या कोई भी हेक्स रंग चुनकर एक कस्टम थीम बनाएं। प्रकाश, अंधेरा और सिस्टम मोड का समर्थन करता है।--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -टैब के साथ व्यापक सेटिंग पैनल: +Comprehensive settings panel with tabs: --**सामान्य**- सिस्टम स्टोरेज, बैकअप प्रबंधन (निर्यात/आयात डेटाबेस) -**प्रकटन**- थीम चयनकर्ता (गहरा/प्रकाश/सिस्टम), रंग थीम प्रीसेट और कस्टम रंग, स्वास्थ्य लॉग दृश्यता, साइडबार आइटम दृश्यता नियंत्रण -**सुरक्षा**- एपीआई एंडपॉइंट सुरक्षा, कस्टम प्रदाता अवरोधन, आईपी फ़िल्टरिंग, सत्र जानकारी -**रूटिंग**- मॉडल उपनाम, पृष्ठभूमि कार्य गिरावट -**लचीलापन**- दर सीमा दृढ़ता, सर्किट ब्रेकर ट्यूनिंग, प्रतिबंधित खातों को स्वचालित रूप से अक्षम करें, प्रदाता समाप्ति की निगरानी -**उन्नत**- कॉन्फ़िगरेशन ओवरराइड, कॉन्फ़िगरेशन ऑडिट ट्रेल, फ़ॉलबैक डिग्रेडेशन मोड![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -एआई कोडिंग टूल के लिए एक-क्लिक कॉन्फ़िगरेशन: क्लाउड कोड, कोडेक्स सीएलआई, जेमिनी सीएलआई, ओपनक्लाव, किलो कोड, एंटीग्रेविटी, क्लाइन, कंटिन्यू, कर्सर और फैक्ट्री ड्रॉयड। सुविधाएँ स्वचालित कॉन्फ़िगरेशन लागू/रीसेट, कनेक्शन प्रोफ़ाइल और मॉडल मैपिंग।![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -सीएलआई एजेंटों की खोज और प्रबंधन के लिए डैशबोर्ड। 14 अंतर्निहित एजेंटों (कोडेक्स, क्लाउड, गूज़, जेमिनी सीएलआई, ओपनक्लाव, एडर, ओपनकोड, क्लाइन, क्वेन कोड, फोर्जकोड, अमेज़ॅन क्यू, ओपन इंटरप्रेटर, कर्सर सीएलआई, वार्प) का ग्रिड दिखाता है: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**इंस्टॉलेशन स्थिति**- संस्करण पहचान के साथ स्थापित / नहीं मिला -**प्रोटोकॉल बैज**- stdio, HTTP, आदि। -**कस्टम एजेंट**- फॉर्म के माध्यम से किसी भी सीएलआई टूल को पंजीकृत करें (नाम, बाइनरी, वर्जन कमांड, स्पॉन आर्ग्स) -**सीएलआई फ़िंगरप्रिंट मिलान**- मूल सीएलआई अनुरोध हस्ताक्षरों से मिलान करने के लिए प्रति-प्रदाता टॉगल करता है, प्रॉक्सी आईपी को संरक्षित करते हुए प्रतिबंध जोखिम को कम करता है--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -डैशबोर्ड से चित्र, वीडियो और संगीत उत्पन्न करें। OpenAI, xAI, टुगेदर, हाइपरबोलिक, SD WebUI, ComfyUI, AnimateDiff, स्टेबल ऑडियो ओपन और MusicGen को सपोर्ट करता है।--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -प्रदाता, मॉडल, खाता और एपीआई कुंजी द्वारा फ़िल्टरिंग के साथ वास्तविक समय अनुरोध लॉगिंग। स्थिति कोड, टोकन उपयोग, विलंबता और प्रतिक्रिया विवरण दिखाता है।![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -क्षमता विश्लेषण के साथ आपका एकीकृत एपीआई एंडपॉइंट: चैट पूर्णताएं, प्रतिक्रिया एपीआई, एंबेडिंग, छवि निर्माण, रीरैंकिंग, ऑडियो ट्रांसक्रिप्शन, टेक्स्ट-टू-स्पीच, मॉडरेशन और पंजीकृत एपीआई कुंजी। रिमोट एक्सेस के लिए क्लाउडफ्लेयर क्विक टनल इंटीग्रेशन और क्लाउड प्रॉक्सी सपोर्ट।![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -एपीआई कुंजी बनाएं, दायरा बढ़ाएं और निरस्त करें। प्रत्येक कुंजी को पूर्ण पहुंच या केवल-पढ़ने की अनुमति वाले विशिष्ट मॉडल/प्रदाताओं तक सीमित किया जा सकता है। उपयोग ट्रैकिंग के साथ विज़ुअल कुंजी प्रबंधन।--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -कार्रवाई प्रकार, अभिनेता, लक्ष्य, आईपी पता और टाइमस्टैम्प द्वारा फ़िल्टरिंग के साथ प्रशासनिक कार्रवाई ट्रैकिंग। पूर्ण सुरक्षा घटना इतिहास.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -विंडोज़, मैकओएस और लिनक्स के लिए नेटिव इलेक्ट्रॉन डेस्कटॉप ऐप। सिस्टम ट्रे एकीकरण, ऑफ़लाइन समर्थन, ऑटो-अपडेट और एक-क्लिक इंस्टॉल के साथ ओमनीरूट को एक स्टैंडअलोन एप्लिकेशन के रूप में चलाएं। +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -मुख्य विशेषताएं: +Key features: -- सर्वर तत्परता मतदान (कोल्ड स्टार्ट पर कोई खाली स्क्रीन नहीं) -- पोर्ट प्रबंधन के साथ सिस्टम ट्रे -- सामग्री सुरक्षा नीति -- सिंगल-इंस्टेंस लॉक -- पुनरारंभ पर स्वतः अद्यतन -- प्लेटफ़ॉर्म-सशर्त यूआई (मैकओएस ट्रैफिक लाइट, विंडोज़/लिनक्स डिफ़ॉल्ट टाइटलबार) -- कठोर इलेक्ट्रॉन बिल्ड पैकेजिंग - स्टैंडअलोन बंडल में सिम्लिंक्ड `नोड_मॉड्यूल` का पता लगाया जाता है और पैकेजिंग से पहले खारिज कर दिया जाता है, जिससे बिल्ड मशीन पर रनटाइम निर्भरता को रोका जा सकता है (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 संपूर्ण दस्तावेज़ीकरण के लिए [`electron/README.md`](../electron/README.md) देखें। +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/in/docs/TROUBLESHOOTING.md b/docs/i18n/in/docs/TROUBLESHOOTING.md index 176bde62dc..de6109cf64 100644 --- a/docs/i18n/in/docs/TROUBLESHOOTING.md +++ b/docs/i18n/in/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -ओम्निरूट के लिए सामान्य समस्याएं और समाधान।--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| समस्या | समाधान | -| ------------------------------------- | ------------------------------------------------------------------------------- | --- | -| पहला लॉगिन काम नहीं कर रहा | `INITIAL_PASSWORD` को `.env` में सेट करें (कोई हार्डकोडेड डिफ़ॉल्ट नहीं) | -| गलत पोर्ट पर डैशबोर्ड खुलता है | `PORT=20128` और `NEXT_PUBLIC_BASE_URL=http://localhost:20128` सेट करें | -| `लॉग/` के अंतर्गत कोई अनुरोध लॉग नहीं | `ENABLE_REQUEST_LOGS=true` सेट करें | -| EACCES: अनुमति अस्वीकृत | `~/.omniroute` को ओवरराइड करने के लिए `DATA_DIR=/path/to/writable/dir` सेट करें | -| रूटिंग रणनीति सहेजी नहीं जा रही | v1.4.11+ पर अपडेट करें (सेटिंग्स दृढ़ता के लिए ज़ोड स्कीमा फिक्स) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**कारण:**प्रदाता कोटा समाप्त हो गया। +**Cause:** Provider quota exhausted. -**ठीक करें:** +**Fix:** -1. डैशबोर्ड कोटा ट्रैकर की जाँच करें -2. फ़ॉलबैक टियर वाले कॉम्बो का उपयोग करें -3. सस्ते/मुफ़्त स्तर पर स्विच करें### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**कारण:**सदस्यता कोटा समाप्त हो गया। +### Rate Limiting -**ठीक करें:** +**Cause:** Subscription quota exhausted. -- फ़ॉलबैक जोड़ें: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- सस्ते बैकअप के रूप में GLM/MiniMax का उपयोग करें### OAuth Token Expired +**Fix:** -ओम्निरूट स्वचालित रूप से टोकन ताज़ा करता है। यदि समस्याएँ बनी रहती हैं: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. डैशबोर्ड → प्रदाता → पुनः कनेक्ट करें -2. प्रदाता कनेक्शन हटाएं और पुनः जोड़ें--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. अपने चल रहे उदाहरण के लिए `BASE_URL` बिंदुओं को सत्यापित करें (उदाहरण के लिए, `http://localhost:20128`) -2. अपने क्लाउड एंडपॉइंट पर `CLOUD_URL` बिंदुओं को सत्यापित करें (उदाहरण के लिए, `https://omniroute.dev`) -3. `NEXT_PUBLIC_*` मानों को सर्वर-साइड मानों के साथ संरेखित रखें### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**लक्षण:**गैर-स्ट्रीमिंग कॉल के लिए क्लाउड एंडपॉइंट पर `अप्रत्याशित टोकन 'डी'...`। +### Cloud `stream=false` Returns 500 -**कारण:**अपस्ट्रीम एसएसई पेलोड लौटाता है जबकि ग्राहक JSON की अपेक्षा करता है। +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**समाधान:**क्लाउड डायरेक्ट कॉल के लिए `stream=true` का उपयोग करें। स्थानीय रनटाइम में SSE→JSON फ़ॉलबैक शामिल है।### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. स्थानीय डैशबोर्ड से एक नई कुंजी बनाएं (`/api/keys`) -2. क्लाउड सिंक चलाएँ: क्लाउड सक्षम करें → अभी सिंक करें -3. पुरानी/गैर-सिंक की गई कुंजियाँ अभी भी क्लाउड पर `401` लौटा सकती हैं--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. रनटाइम फ़ील्ड जांचें: `कर्ल http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. पोर्टेबल मोड के लिए: छवि लक्ष्य `रनर-सीएलआई` (बंडल सीएलआई) का उपयोग करें -3. होस्ट माउंट मोड के लिए: `CLI_EXTRA_PATHS` सेट करें और होस्ट बिन निर्देशिका को केवल पढ़ने के लिए माउंट करें -4. यदि `इंस्टॉल = सही` और `रनने योग्य = गलत`: बाइनरी पाया गया था लेकिन स्वास्थ्य जांच विफल रही### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. डैशबोर्ड → उपयोग में उपयोग के आँकड़े जाँचें -2. प्राथमिक मॉडल को जीएलएम/मिनीमैक्स पर स्विच करें -3. गैर-महत्वपूर्ण कार्यों के लिए फ्री टियर (मिथुन सीएलआई, क्यूडर) का उपयोग करें -4. प्रति एपीआई कुंजी लागत बजट निर्धारित करें: डैशबोर्ड → एपीआई कुंजी → बजट--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -अपनी `.env` फ़ाइल में `ENABLE_REQUEST_LOGS=true` सेट करें। लॉग `लॉग/` निर्देशिका के अंतर्गत दिखाई देते हैं।### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,98 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- मुख्य स्थिति: `${DATA_DIR}/storage.sqlite` (प्रदाता, कॉम्बो, उपनाम, कुंजियाँ, सेटिंग्स) -- उपयोग: `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) में SQLite टेबल + वैकल्पिक `${DATA_DIR}/log.txt` और `${DATA_DIR}/call_logs/` -- अनुरोध लॉग: `/logs/...` (जब `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -जब किसी प्रदाता का सर्किट ब्रेकर खुला होता है, तो कूलडाउन समाप्त होने तक अनुरोध अवरुद्ध हो जाते हैं। +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**ठीक करें:** +**Fix:** -1.**डैशबोर्ड → सेटिंग्स → लचीलापन**पर जाएं 2. प्रभावित प्रदाता के लिए सर्किट ब्रेकर कार्ड की जाँच करें 3. सभी ब्रेकर साफ़ करने के लिए**रीसेट ऑल**पर क्लिक करें, या कूलडाउन समाप्त होने तक प्रतीक्षा करें 4. रीसेट करने से पहले सत्यापित करें कि प्रदाता वास्तव में उपलब्ध है### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -यदि कोई प्रदाता बार-बार खुली स्थिति में प्रवेश करता है: +### Provider keeps tripping the circuit breaker -1. विफलता पैटर्न के लिए**डैशबोर्ड → स्वास्थ्य → प्रदाता स्वास्थ्य**की जाँच करें 2.**सेटिंग्स → लचीलापन → प्रदाता प्रोफाइल**पर जाएं और विफलता सीमा बढ़ाएं -2. जांचें कि क्या प्रदाता ने एपीआई सीमाएं बदल दी हैं या पुनः प्रमाणीकरण की आवश्यकता है -3. विलंबता टेलीमेट्री की समीक्षा करें - उच्च विलंबता टाइमआउट-आधारित विफलताओं का कारण बन सकती है--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- सुनिश्चित करें कि आप सही उपसर्ग का उपयोग कर रहे हैं: `डीपग्राम/नोवा-3` या `असेंबलीई/बेस्ट` -- सत्यापित करें कि प्रदाता**डैशबोर्ड → प्रदाता**में जुड़ा हुआ है### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- समर्थित ऑडियो प्रारूप जांचें: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- सत्यापित करें कि फ़ाइल का आकार प्रदाता सीमा के भीतर है (आमतौर पर <25MB) -- प्रदाता कार्ड में प्रदाता एपीआई कुंजी वैधता की जांच करें--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -प्रारूप अनुवाद समस्याओं को डीबग करने के लिए**डैशबोर्ड → अनुवादक**का उपयोग करें: +Use **Dashboard → Translator** to debug format translation issues: -| मोड | कब उपयोग करें | -| ---------------- | ------------------------------------------------------------------------------------------------------------- | ------------------------ | -| **खेल का मैदान** | इनपुट/आउटपुट स्वरूपों की साथ-साथ तुलना करें - यह कैसे अनुवादित होता है यह देखने के लिए एक असफल अनुरोध चिपकाएँ | -| **चैट परीक्षक** | लाइव संदेश भेजें और हेडर सहित पूर्ण अनुरोध/प्रतिक्रिया पेलोड का निरीक्षण करें | -| **टेस्ट बेंच** | यह पता लगाने के लिए कि कौन से अनुवाद टूटे हुए हैं, सभी प्रारूप संयोजनों में बैच परीक्षण चलाएँ | -| **लाइव मॉनिटर** | रुक-रुक कर होने वाली अनुवाद समस्याओं को पकड़ने के लिए वास्तविक समय अनुरोध प्रवाह देखें | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**सोच टैग दिखाई नहीं दे रहे हैं**- जांचें कि क्या लक्ष्य प्रदाता सोच और सोच बजट सेटिंग का समर्थन करता है -**टूल कॉल ड्रॉपिंग**— कुछ प्रारूप अनुवाद असमर्थित फ़ील्ड को हटा सकते हैं; खेल का मैदान मोड में सत्यापित करें -**सिस्टम प्रॉम्प्ट गायब**— क्लाउड और जेमिनी हैंडल सिस्टम प्रॉम्प्ट अलग-अलग होते हैं; अनुवाद आउटपुट की जाँच करें -**एसडीके ऑब्जेक्ट के बजाय कच्ची स्ट्रिंग लौटाता है**- v1.1.0 में फिक्स्ड: रिस्पॉन्स सैनिटाइज़र अब गैर-मानक फ़ील्ड (`x_groq`, `usage_breakdown`, आदि) को हटा देता है जो OpenAI SDK पायडेंटिक सत्यापन विफलताओं का कारण बनता है -**GLM/ERNIE `सिस्टम' भूमिका को अस्वीकार करता है**- v1.1.0 में फिक्स्ड: रोल नॉर्मलाइज़र स्वचालित रूप से असंगत मॉडल के लिए सिस्टम संदेशों को उपयोगकर्ता संदेशों में मर्ज कर देता है -**'डेवलपर' की भूमिका पहचानी नहीं गई**- v1.1.0 में फिक्स्ड: गैर-ओपनएआई प्रदाताओं के लिए स्वचालित रूप से `सिस्टम' में कनवर्ट किया गया --**`json_schema`जेमिनी के साथ काम नहीं कर रहा है**- v1.1.0 में फिक्स्ड:`response_format`को अब जेमिनी के`responseMimeType`+`responseSchema` में बदल दिया गया है--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- ऑटो दर-सीमा केवल एपीआई कुंजी प्रदाताओं पर लागू होती है (OAuth/सदस्यता पर नहीं) -- सत्यापित करें**सेटिंग्स → लचीलापन → प्रदाता प्रोफाइल**में ऑटो-दर-सीमा सक्षम है -- जांचें कि क्या प्रदाता `429` स्टेटस कोड या `रीट्री-आफ्टर` हेडर लौटाता है### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -प्रदाता प्रोफ़ाइल इन सेटिंग्स का समर्थन करती हैं: +### Tuning exponential backoff --**आधार विलंब**— पहली विफलता के बाद प्रारंभिक प्रतीक्षा समय (डिफ़ॉल्ट: 1 सेकंड) -**अधिकतम विलंब**— अधिकतम प्रतीक्षा समय सीमा (डिफ़ॉल्ट: 30s) -**गुणक**- लगातार विफलता के बाद विलंब को कितना बढ़ाया जाए (डिफ़ॉल्ट: 2x)### Anti-thundering herd +Provider profiles support these settings: -जब कई समवर्ती अनुरोध एक दर-सीमित प्रदाता से टकराते हैं, तो ओमनीरूट अनुरोधों को क्रमबद्ध करने और कैस्केडिंग विफलताओं को रोकने के लिए म्यूटेक्स + ऑटो रेट-लिमिटिंग का उपयोग करता है। यह एपीआई कुंजी प्रदाताओं के लिए स्वचालित है।--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -कुछ ओमनीरूट उपयोगकर्ता गेटवे को RAG या एजेंट स्टैक के सामने रखते हैं। उन सेटअपों में एक अजीब पैटर्न देखना आम है: ओम्नीरूट स्वस्थ दिखता है (प्रदाता ऊपर, रूटिंग प्रोफाइल ठीक, कोई दर सीमा अलर्ट नहीं) लेकिन अंतिम उत्तर अभी भी गलत है। +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -व्यवहार में ये घटनाएं आम तौर पर डाउनस्ट्रीम आरएजी पाइपलाइन से आती हैं, गेटवे से नहीं। +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -यदि आप उन विफलताओं का वर्णन करने के लिए एक साझा शब्दावली चाहते हैं तो आप डब्लूएफजीवाई प्रॉब्लममैप का उपयोग कर सकते हैं, एक बाहरी एमआईटी लाइसेंस टेक्स्ट संसाधन जो सोलह आवर्ती आरएजी / एलएलएम विफलता पैटर्न को परिभाषित करता है। उच्च स्तर पर इसमें शामिल हैं: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- पुनर्प्राप्ति बहाव और टूटी हुई संदर्भ सीमाएँ -- खाली या बासी इंडेक्स और वेक्टर स्टोर -- एम्बेडिंग बनाम सिमेंटिक बेमेल -- शीघ्र असेंबली और संदर्भ विंडो समस्याएँ -- तर्क पतन और अतिआत्मविश्वासपूर्ण उत्तर -- लंबी श्रृंखला और एजेंट समन्वय विफलताएँ -- मल्टी एजेंट मेमोरी और रोल ड्रिफ्ट -- परिनियोजन और बूटस्ट्रैप ऑर्डरिंग समस्याएं +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -विचार सरल है: +The idea is simple: -1. जब आप किसी खराब प्रतिक्रिया की जांच करते हैं, तो कैप्चर करें: - - उपयोगकर्ता कार्य और अनुरोध - - ओमनीरूट में रूट या प्रदाता कॉम्बो - - डाउनस्ट्रीम में उपयोग किया गया कोई भी RAG संदर्भ (पुनर्प्राप्त दस्तावेज़, टूल कॉल, आदि) -2. घटना को एक या दो WFGY समस्या मानचित्र संख्याओं ('नंबर 1' ... 'नंबर 16') पर मैप करें। -3. नंबर को अपने डैशबोर्ड, रनबुक, या घटना ट्रैकर में ओमनीरूट लॉग के बगल में संग्रहीत करें। -4. यह तय करने के लिए संबंधित WFGY पृष्ठ का उपयोग करें कि आपको अपने RAG स्टैक, रिट्रीवर या रूटिंग रणनीति को बदलने की आवश्यकता है या नहीं। +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -पूर्ण पाठ और ठोस व्यंजन यहां उपलब्ध हैं (एमआईटी लाइसेंस, केवल पाठ): +Full text and concrete recipes live here (MIT license, text only): -[डब्ल्यूएफजीवाई प्रॉब्लममैप रीडमी](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) +[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -यदि आप ओमनीरूट के पीछे आरएजी या एजेंट पाइपलाइन नहीं चलाते हैं तो आप इस अनुभाग को अनदेखा कर सकते हैं।--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**गिटहब मुद्दे**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**आर्किटेक्चर**: आंतरिक विवरण के लिए [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) देखें -**एपीआई संदर्भ**: सभी समापन बिंदुओं के लिए [`docs/API_REFERENCE.md`](API_REFERENCE.md) देखें -**स्वास्थ्य डैशबोर्ड**: वास्तविक समय प्रणाली की स्थिति के लिए**डैशबोर्ड → स्वास्थ्य**जांचें -**अनुवादक**: प्रारूप संबंधी समस्याओं को डीबग करने के लिए**डैशबोर्ड → अनुवादक**का उपयोग करें +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/in/llm.txt b/docs/i18n/in/llm.txt new file mode 100644 index 0000000000..c63b854e90 --- /dev/null +++ b/docs/i18n/in/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (हिन्दी (IN)) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## अवलोकन + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### सुरक्षा +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/it/README.md b/docs/i18n/it/README.md index 846a8d388c..6580fe5de9 100644 --- a/docs/i18n/it/README.md +++ b/docs/i18n/it/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Il tuo proxy API universale: un endpoint, oltre 60 provider, zero tempi di inattività. Ora con**Server MCP (25 strumenti)**,**Protocollo A2A**,**Sistemi di memoria/competenze**e**App desktop Electron**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Completamenti chat • Incorporamenti • Generazione di immagini • Video • Musica • Audio • Riclassificazione •**Ricerca Web**• Server MCP • Protocollo A2A • 100% TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Il tuo proxy API universale: un endpoint, oltre 60 provider, zero tempi di inat [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Sito web](https://omniroute.online) • [🚀 Avvio rapido](#-quick-start) • [💡 Funzionalità](#-caratteristiche-chiave) • [📖 Documenti](#-documentazione) • [💰 Prezzi](#-prezzi-in-un-colpo) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Available in:**🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasile)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Spagnolo](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portogallo)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filippino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,549 +60,623 @@ _Il tuo proxy API universale: un endpoint, oltre 60 provider, zero tempi di inat ## 📸 Dashboard Preview - -Fai clic per visualizzare gli screenshot della dashboard +
+Click to see dashboard screenshots -| Pagina | Schermata | -| ------------------------ | --------------------------------------------------- | ---------- | -| **Fornitori** | ![Providers](docs/screenshots/01-provviders.png) | -| **Combo** | ![Combos](docs/screenshots/02-combos.png) | -| **Analisi** | ![Analisi](docs/screenshots/03-analytics.png) | -| **Salute** | ![Salute](docs/screenshots/04-health.png) | -| **Traduttore** | ![Traduttore](docs/screenshots/05-translator.png) | -| **Impostazioni** | ![Impostazioni](docs/screenshots/06-settings.png) | -| **Strumenti CLI** | ![Strumenti CLI](docs/screenshots/07-cli-tools.png) | -| **Registri di utilizzo** | ![Utilizzo](docs/screenshots/08-usage.png) | -| **Endpoint** | ![Endpoint](docs/screenshots/09-endpoint.png) |
| +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | + + --- ### 🤖 Free AI Provider for your favorite coding agents -_Connetti qualsiasi strumento IDE o CLI basato sull'intelligenza artificiale tramite OmniRoute: gateway API gratuito per codifica illimitata._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ - + - - - - - - - - - - - +
+ OpenClaw
OpenClaw

⭐ 205K
+ - NanoBot
+ NanoBot
NanoBot

- ⭐ 20,9K + ⭐ 20.9K
+ - PicoClaw
+ PicoClaw
PicoClaw

- ⭐ 14,6K + ⭐ 14.6K
+ ZeroClaw
ZeroClaw

- ⭐ 9,9K + ⭐ 9.9K
+ IronClaw
- Artiglio di Ferro + IronClaw

- ⭐ 2,1K + ⭐ 2.1K
+ OpenCode
OpenCode

⭐ 106K
+ Codex CLI
- CLI del Codice + Codex CLI

- ⭐ 60,8K + ⭐ 60.8K
+ - Codice Claude
- Codice Claude + Claude Code
+ Claude Code

- ⭐ 67,3K + ⭐ 67.3K
+ Gemini CLI
- CLI Gemini + Gemini CLI

- ⭐ 94,7K + ⭐ 94.7K
+ - Codice Kilo
- Codice chilo + Kilo Code
+ Kilo Code

- ⭐ 15,5K + ⭐ 15.5K
-📡 Tutti gli agenti si connettono tramite http://localhost:20128/v1 o http://cloud.omniroute.online/v1: una configurazione, modelli e quote illimitati--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Smetti di sprecare denaro e di superare i limiti:** +**Stop wasting money and hitting limits:** -- La quota di abbonamento scade ogni mese inutilizzata +- Subscription quota expires unused every month - Rate limits stop you mid-coding -- API costose ($ 20-50/mese per fornitore) -- Passaggio manuale tra fornitori +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute risolve questo problema:** +**OmniRoute solves this:** -- ✅**Massimizza gli abbonamenti**- Tieni traccia della quota, utilizza ogni bit prima di reimpostarlo -- ✅**Falback automatico**- Abbonamento → Chiave API → Economico → Gratuito, zero tempi di inattività -- ✅**Multi-account**- Round robin tra account per fornitore -- ✅**Universale**- Funziona con Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, qualsiasi strumento CLI--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Join our community!**[WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Sito web**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Problemi**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Gruppo della community](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Contributing**: vedi [CONTRIBUTING.md](CONTRIBUTING.md), apri un PR o scegli un "buon primo numero" -**Progetto originale**: [9router di decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Quando apri un problema, esegui il comando system-info e allega il file generato:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Questo genera un `system-info.txt` con la versione di Node.js, la versione di OmniRoute, i dettagli del sistema operativo, gli strumenti CLI installati (qoder, gemini, claude, codex, antigravity, droid, ecc.), lo stato di Docker/PM2 e i pacchetti di sistema: tutto ciò di cui abbiamo bisogno per riprodurre rapidamente il problema. Allega il file direttamente al tuo problema GitHub.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Ogni sviluppatore che utilizza strumenti di intelligenza artificiale affronta questi problemi quotidianamente.**OmniRoute è stato creato per risolverli tutti: dai superamenti dei costi ai blocchi regionali, dai flussi OAuth interrotti alle operazioni di protocollo e all'osservabilità aziendale. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. - -💸 1. "Pago un abbonamento costoso ma vengo comunque interrotto dai limiti" +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Gli sviluppatori pagano $ 20-200 al mese per Claude Pro, Codex Pro o GitHub Copilot. Anche pagando, la quota ha un tetto: 5 ore di utilizzo, limiti settimanali o limiti di tariffa al minuto. A metà sessione di codifica, il provider smette di rispondere e lo sviluppatore perde flusso e produttività. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Come OmniRoute risolve il problema:** +**How OmniRoute solves it:** --**Fallback intelligente a 4 livelli**: se la quota dell'abbonamento si esaurisce, reindirizza automaticamente alla chiave API → Economico → Gratuito senza alcun intervento manuale --**Tracciamento dei limiti del provider**: gli snapshot delle quote memorizzate nella cache si aggiornano in base a una pianificazione lato server (predefinita `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) con aggiornamento manuale disponibile nell'interfaccia utente --**Supporto multi-account**: più account per fornitore con round robin automatico: quando uno si esaurisce, passa a quello successivo --**Combo personalizzate**— Catene di fallback personalizzabili con 9 strategie di bilanciamento (priorità, ponderata, riempimento-prima, round-robin, P2C, casuale, meno utilizzata, ottimizzata in termini di costi, strettamente casuale) --**Quote aziendali Codex**: monitoraggio delle quote dello spazio di lavoro aziendale/team direttamente nella dashboard
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard - -🔌 2. "Devo utilizzare più provider, ma ognuno ha un'API diversa" + -OpenAI utilizza un formato, Claude (Anthropic) ne utilizza un altro, Gemini ancora un altro. Se uno sviluppatore desidera testare modelli di fornitori diversi o eseguire il fallback tra di loro, deve riconfigurare gli SDK, modificare gli endpoint e gestire formati incompatibili. I provider personalizzati (FriendLI, NIM) hanno endpoint del modello non standard. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Come OmniRoute risolve il problema:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Endpoint unificato**: un singolo `http://localhost:20128/v1` funge da proxy per tutti gli oltre 60 provider --**Format Translation**— Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API --**Sanitizzazione della risposta**: rimuove i campi non standard (`x_groq`, `usage_breakdown`, `service_tier`) che interrompono OpenAI SDK v1.83+ --**Normalizzazione dei ruoli**: converte `developer` → `system` per provider non OpenAI; `sistema` → `utente` per GLM/ERNIE --**Think Tag Extraction**— Estrae i blocchi "" da modelli come DeepSeek R1 in "reasoning_content" standardizzato --**Output strutturato per Gemini**— Conversione automatica `json_schema` → `responseMimeType`/`responseSchema` --**`stream` è impostato su `false`**— Si allinea con le specifiche OpenAI, evitando SSE imprevisti negli SDK Python/Rust/Go
+**How OmniRoute solves it:** - -🌐 3. "Il mio provider di intelligenza artificiale blocca la mia regione/paese" +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Provider come OpenAI/Codex bloccano l'accesso da determinate regioni geografiche. Gli utenti ricevono errori come "unsupported_country_region_territory" durante le connessioni OAuth e API. Ciò è particolarmente frustrante per gli sviluppatori dei paesi in via di sviluppo. + -**Come OmniRoute risolve il problema:** +
+🌐 3. "My AI provider blocks my region/country" --**Configurazione proxy a 3 livelli**: proxy configurabile a 3 livelli: globale (tutto il traffico), per provider (un solo provider) e per connessione/chiave --**Badge proxy con codice colore**— Indicatori visivi: 🟢 proxy globale, 🟡 proxy provider, 🔵 proxy di connessione, che mostra sempre l'IP --**Scambio di token OAuth tramite proxy**: anche il flusso OAuth passa attraverso il proxy, risolvendo il problema `unsupported_country_region_territory` --**Test di connessione tramite proxy**: i test di connessione utilizzano il proxy configurato (non più bypass diretto) --**Supporto SOCKS5**: supporto completo del proxy SOCKS5 per il routing in uscita --**TLS Fingerprint Spoofing**: impronta digitale TLS simile a un browser tramite `wreq-js` per bypassare il rilevamento dei bot --**🔏 Corrispondenza impronta digitale CLI**: riordina le intestazioni e i campi del corpo in modo che corrispondano alle firme binarie native della CLI, riducendo drasticamente il rischio di segnalazione dell'account. L'IP proxy viene preservato: ottieni contemporaneamente il mascheramento IP stealth**e**
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. - -🆓 4. "Voglio usare l'intelligenza artificiale per programmare ma non ho soldi" +**How OmniRoute solves it:** -Non tutti possono pagare $ 20-200 al mese per gli abbonamenti AI. Studenti, sviluppatori provenienti da paesi emergenti, hobbisti e liberi professionisti hanno bisogno di accedere a modelli di qualità a costo zero. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Come OmniRoute risolve il problema:** + --**Provider di livello gratuito integrati**: supporto nativo per provider gratuiti al 100%: Qoder (5 modelli illimitati tramite OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 modelli illimitati: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + ID AWS Builder gratuiti), Gemini CLI (180.000 token/mese gratuiti) --**Ollama Cloud**: modelli Ollama ospitati nel cloud su `api.ollama.com` con livello "Utilizzo leggero" gratuito; utilizzare il prefisso `ollamacloud/` --**Combo solo gratuiti**— Catena `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $ 0/mese con zero tempi di inattività --**Accesso gratuito a NVIDIA NIM**: ~40 RPM accesso gratuito per sviluppatori a oltre 70 modelli su build.nvidia.com (passaggio dai crediti ai limiti di velocità puri) --**Strategia di ottimizzazione dei costi**: strategia di routing che sceglie automaticamente il fornitore più economico disponibile +
+🆓 4. "I want to use AI for coding but I have no money" - -🔒 5. "Devo proteggere il mio gateway AI da accessi non autorizzati" +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -Quando si espone un gateway AI alla rete (LAN, VPS, Docker), chiunque abbia l'indirizzo può consumare i token/la quota dello sviluppatore. Senza protezione, le API sono vulnerabili ad usi impropri, tempestive iniezioni e abusi. +**How OmniRoute solves it:** -**Come OmniRoute risolve il problema:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**Gestione delle chiavi API**: generazione, rotazione e definizione dell'ambito per provider con una pagina `/dashboard/api-manager` dedicata --**Autorizzazioni a livello di modello**: limita le chiavi API a modelli specifici (`openai/*`, modelli con caratteri jolly), con l'interruttore Consenti tutto/Limita --**API Endpoint Protection**: richiede una chiave per "/v1/models" e blocca fornitori specifici dall'elenco --**Auth Guard + Protezione CSRF**: tutti i percorsi del dashboard protetti con middleware `withAuth` + token CSRF --**Rate Limiter**: limitazione della velocità per IP con finestre configurabili --**IP Filtering**— Allowlist/blocklist for access control --**Prompt Injection Guard**: sanificazione contro modelli di prompt dannosi --**Crittografia AES-256-GCM**: credenziali crittografate a riposo
+ - -🛑 6. "Il mio provider si è interrotto e ho perso il flusso di codifica" +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -I fornitori di intelligenza artificiale possono diventare instabili, restituire errori 5xx o raggiungere limiti di velocità temporanei. Se uno sviluppatore dipende da un singolo fornitore, viene interrotto. Without circuit breakers, repeated retries can crash the application. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Come OmniRoute risolve il problema:** +**How OmniRoute solves it:** --**Interruttore automatico per modello**: apertura/chiusura automatica con soglie e raffreddamento configurabili (chiuso/aperto/semiaperto), con ambito per modello per evitare blocchi a cascata --**Backoff esponenziale**: ritardi progressivi tra i tentativi --**Anti-Thundering Herd**— Mutex + protezione semaforo contro tempeste di tentativi simultanei --**Catene di fallback combinate**: se il fornitore primario fallisce, cade automaticamente nella catena senza alcun intervento --**Combo Circuit Breaker**: disabilita automaticamente i provider in errore all'interno di una catena combinata --**Dashboard integrità**: monitoraggio del tempo di attività, stati degli interruttori automatici, blocchi, statistiche della cache, latenza p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest - -🔧 7. "Configurare ogni strumento AI è noioso e ripetitivo" + -Gli sviluppatori utilizzano Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Ogni strumento necessita di una configurazione diversa (endpoint API, chiave, modello). La riconfigurazione quando si cambia fornitore o modello è una perdita di tempo. +
+🛑 6. "My provider went down and I lost my coding flow" -**Come OmniRoute risolve il problema:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**Dashboard degli strumenti CLI**: pagina dedicata con configurazione con un clic per Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline --**Generatore di configurazione di GitHub Copilot**: genera `chatLanguageModels.json` per VS Code con selezione di modelli in blocco --**Procedura guidata di onboarding**: configurazione guidata in 4 passaggi per gli utenti alle prime armi --**Un endpoint, tutti i modelli**— Configura `http://localhost:20128/v1` una volta, accedi a oltre 60 provider
+**How OmniRoute solves it:** - -🔑 8. "Gestire token OAuth da più provider è un inferno" +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot: utilizzano tutti OAuth 2.0 con token in scadenza. Gli sviluppatori devono autenticarsi nuovamente costantemente, gestire "client_secret is Missing", "redirect_uri_mismatch" e errori sui server remoti. OAuth su LAN/VPS è particolarmente problematico. + -**Come OmniRoute risolve il problema:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Aggiornamento automatico dei token**: i token OAuth si aggiornano in background prima della scadenza --**OAuth 2.0 (PKCE) integrato**: flusso automatico per Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder --**OAuth multi-account**: account multipli per provider tramite estrazione di token JWT/ID --**OAuth LAN/Remote Fix**: rilevamento IP privato per `redirect_uri` + modalità URL manuale per server remoti --**OAuth Behind Nginx**: utilizza `window.location.origin` per la compatibilità con il proxy inverso --**Guida OAuth remota**: guida passo passo per le credenziali Google Cloud su VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. - -📊 9. "Non so quanto sto spendendo né dove" +**How OmniRoute solves it:** -Gli sviluppatori utilizzano più fornitori a pagamento ma non hanno una visione unificata della spesa. Ogni fornitore ha il proprio dashboard di fatturazione, ma non esiste una visualizzazione consolidata. I costi imprevisti possono accumularsi. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Come OmniRoute risolve il problema:** + --**Dashboard di analisi dei costi**: monitoraggio dei costi per token e gestione del budget per fornitore --**Limiti di budget per livello**: massimale di spesa per livello che attiva il fallback automatico --**Configurazione dei prezzi per modello**: prezzi configurabili per modello --**Statistiche di utilizzo per chiave API**: conteggio delle richieste e timestamp dell'ultimo utilizzo per chiave --**Dashboard di analisi**: schede statistiche, grafico di utilizzo del modello, tabella dei fornitori con percentuali di successo e latenza +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" - -🐛 10. "Non riesco a diagnosticare errori e problemi nelle chiamate AI" +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Quando una chiamata fallisce, lo sviluppatore non sa se si trattava di un limite di velocità, di un token scaduto, di un formato errato o di un errore del provider. Registri frammentati su diversi terminali. Senza osservabilità, il debug è un processo per tentativi ed errori. +**How OmniRoute solves it:** -**Come OmniRoute risolve il problema:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Dashboard dei registri unificati**: 4 schede: registri delle richieste, registri del proxy, registri di controllo, console --**Visualizzatore log della console**: visualizzatore in stile terminale in tempo reale con livelli codificati a colori, scorrimento automatico, ricerca, filtro --**Registri proxy SQLite**: registri persistenti che sopravvivono ai riavvii del server --**Translator Playground**— 4 modalità di debug: Playground (traduzione del formato), Chat Tester (andata e ritorno), Test Bench (batch), Live Monitor (in tempo reale) --**Telemetria richiesta**: latenza p50/p95/p99 + traccia X-Request-Id --**Registrazione basata su file con rotazione**: i registri delle app ruotano in base a dimensioni, giorni di conservazione e numero di archivi; gli artefatti del registro chiamate ruotano in base ai giorni di conservazione e al numero di file --**Rapporto informazioni di sistema**: `npm run system-info` genera `system-info.txt` con l'ambiente completo (versione del nodo, versione di OmniRoute, sistema operativo, strumenti CLI, stato Docker/PM2). Allegalo quando segnali problemi per il triage immediato.
+ - -🏗️ 11. "L'implementazione e la manutenzione del gateway sono complesse" +
+📊 9. "I don't know how much I'm spending or where" -L'installazione, la configurazione e la manutenzione di un proxy AI in diversi ambienti (locale, VPS, Docker, cloud) richiedono molto lavoro. Problemi come percorsi hardcoded, "EACCES" nelle directory, conflitti di porte e build multipiattaforma aggiungono attrito. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Come OmniRoute risolve il problema:** +**How OmniRoute solves it:** --**npm global install**— `npm install -g omniroute && omniroute` — fatto --**Docker multipiattaforma**— AMD64 + ARM64 nativo (Apple Silicon, AWS Graviton, Raspberry Pi) --**Docker Compose Profiles**— `base` (nessuno strumento CLI) e `cli` (con Claude Code, Codex, OpenClaw) --**App desktop Electron**: app nativa per Windows/macOS/Linux con barra delle applicazioni, avvio automatico, modalità offline --**Modalità porta divisa**: API e dashboard su porte separate per scenari avanzati (proxy inverso, rete di contenitori) --**Cloud Sync**: configura la sincronizzazione tra dispositivi tramite Cloudflare Workers --**Backup DB**: backup, ripristino, esportazione e importazione automatici di tutte le impostazioni, con `DISABLE_SQLITE_AUTO_BACKUP` per backup gestiti esternamente
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency - -🌍 12. "L'interfaccia è solo in inglese e il mio team non parla inglese" + -I team nei paesi non anglofoni, soprattutto in America Latina, Asia ed Europa, hanno difficoltà con le interfacce solo in inglese. Le barriere linguistiche riducono l'adozione e aumentano gli errori di configurazione. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Come OmniRoute risolve il problema:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Dashboard i18n — 30 lingue**— Tutti gli oltre 500 tasti tradotti tra cui arabo, bulgaro, danese, tedesco, spagnolo, finlandese, francese, ebraico, hindi, ungherese, indonesiano, italiano, giapponese, coreano, malese, olandese, norvegese, polacco, portoghese (PT/BR), rumeno, russo, slovacco, svedese, tailandese, ucraino, vietnamita, cinese, filippino, inglese --**Supporto RTL**: supporto da destra a sinistra per arabo ed ebraico --**README multilingue**: 30 traduzioni complete di documentazione --**Selettore lingua**: icona del globo nell'intestazione per la commutazione in tempo reale
+**How OmniRoute solves it:** - -🔄 13. "Mi serve qualcosa di più della semplice chat: mi servono incorporamenti, immagini e audio" +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -L'intelligenza artificiale non è solo il completamento della chat. Gli sviluppatori devono generare immagini, trascrivere audio, creare incorporamenti per RAG, riclassificare i documenti e moderare i contenuti. Ogni API ha un endpoint e un formato diversi. + -**Come OmniRoute risolve il problema:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Embedding**— `/v1/embeddings` con 6 provider e oltre 9 modelli --**Generazione di immagini**— `/v1/images/ generations` con 10 provider e oltre 20 modelli (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Trasformazione testo in video**— `/v1/videos/generazioni` — ComfyUI (AnimateDiff, SVD) e SD WebUI --**Trasformazione testo in musica**— `/v1/music/ generations` — ComfyUI (Stable Audio Open, MusicGen) --**Trascrizione audio**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Sintesi vocale**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + fornitori esistenti --**Moderazioni**— `/v1/moderations` — Controlli di sicurezza dei contenuti --**Riclassificazione**— `/v1/rerank` — Riclassificazione della pertinenza del documento --**API Response**: supporto completo `/v1/responses` per Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. - -🧪 14. "Non ho modo di testare e confrontare la qualità dei modelli" +**How OmniRoute solves it:** -Gli sviluppatori vogliono sapere quale modello è il migliore per il loro caso d'uso (codice, traduzione, ragionamento), ma il confronto manuale è lento. Non esistono strumenti di valutazione integrati. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**Come OmniRoute risolve il problema:** + --**Valutazioni LLM**: test Golden Set con 10 casi precaricati che coprono saluti, matematica, geografia, generazione di codice, conformità JSON, traduzione, ribasso, rifiuto di sicurezza --**4 strategie di corrispondenza**: "esatto", "contiene", "regex", "personalizzato" (funzione JS) --**Translator Playground Test Bench**: test in batch con input multipli e output previsti, confronto tra provider --**Chat Tester**: andata e ritorno completo con rendering della risposta visiva --**Live Monitor**: flusso in tempo reale di tutte le richieste che passano attraverso il proxy +
+🌍 12. "The interface is English-only and my team doesn't speak English" - -📈 15. "Ho bisogno di scalare senza perdere prestazioni" +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -Man mano che il volume delle richieste cresce, senza la memorizzazione nella cache le stesse domande generano costi duplicati. Senza idempotenza, le richieste duplicate sprecano elaborazione. I limiti tariffari per fornitore devono essere rispettati. +**How OmniRoute solves it:** -**Come OmniRoute risolve il problema:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Cache semantica**: la cache a due livelli (firma + semantica) riduce costi e latenza --**Idempotenza richiesta**: finestra di deduplicazione di 5 secondi per richieste identiche --**Rilevamento del limite di velocità**: RPM per provider, gap minimo e monitoraggio simultaneo massimo --**Limiti di velocità modificabili**: impostazioni predefinite configurabili in Impostazioni → Resilienza con persistenza --**Cache di convalida della chiave API**: cache a 3 livelli per prestazioni di produzione --**Dashboard integrità con telemetria**: latenza p50/p95/p99, statistiche cache, tempo di attività
+ - -🤖 16. "Voglio controllare il comportamento del modello a livello globale" +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Sviluppatori che desiderano tutte le risposte in una lingua specifica, con un tono specifico o che desiderano limitare i token di ragionamento. Configurarlo in ogni strumento/richiesta non è pratico. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Come OmniRoute risolve il problema:** +**How OmniRoute solves it:** --**Inserimento prompt di sistema**: prompt globale applicato a tutte le richieste --**Thinking Budget Validation**: controllo dell'allocazione dei token tramite ragionamento per richiesta (passthrough, automatico, personalizzato, adattivo) --**9 Strategie di routing**: strategie globali che determinano la modalità di distribuzione delle richieste --**Wildcard Router**— I modelli `provider/*` instradano dinamicamente a qualsiasi provider --**Abilita/Disabilita combo**: attiva/disattiva le combo direttamente dalla dashboard --**Attiva/disattiva provider**: attiva/disattiva tutte le connessioni per un provider con un clic --**Provider bloccati**— Esclude fornitori specifici dall'elenco "/v1/models".
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex - -🧰 17. "Ho bisogno degli strumenti MCP come funzionalità di prodotto di prima classe" + -Molti gateway AI espongono MCP solo come dettaglio di implementazione nascosto. I team hanno bisogno di un livello operativo visibile e gestibile. +
+🧪 14. "I have no way to test and compare quality across models" -**Come OmniRoute risolve il problema:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP viene visualizzato nella navigazione del dashboard e nella scheda del protocollo dell'endpoint -- Pagina di gestione MCP dedicata con processo, strumenti, ambiti e audit -- Avvio rapido integrato per `omniroute --mcp` e onboarding del client
+**How OmniRoute solves it:** - -🧠 18. "Ho bisogno dell'orchestrazione A2A con percorsi di attività di sincronizzazione e streaming" +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -I flussi di lavoro degli agenti necessitano sia di risposte dirette che di esecuzione in streaming di lunga durata con controllo del ciclo di vita. + -**Come OmniRoute risolve il problema:** +
+📈 15. "I need to scale without losing performance" -- Endpoint A2A JSON-RPC (`POST /a2a`) con "messaggio/invia" e "messaggio/stream" -- Streaming SSE con propagazione dello stato terminale -- API del ciclo di vita delle attività per "tasks/get" e "tasks/cancel".
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. - -🛰️ 19. "Ho bisogno dello stato reale del processo MCP, non di uno stato indovinato" +**How OmniRoute solves it:** -I team operativi devono sapere se MCP è effettivamente attivo, non solo se un'API è raggiungibile. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**Come OmniRoute risolve il problema:** + -- File heartbeat di runtime con PID, timestamp, trasporto, conteggio strumenti e modalità ambito -- API di stato MCP che combina battito cardiaco + attività recente -- Schede di stato dell'interfaccia utente per l'aggiornamento di processo/tempo di attività/battito cardiaco +
+🤖 16. "I want to control model behavior globally" - -📋 20. "Ho bisogno dell'esecuzione verificabile dello strumento MCP" +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Quando gli strumenti modificano la configurazione o attivano azioni operative, i team necessitano di tracciabilità forense. +**How OmniRoute solves it:** -**Come OmniRoute risolve il problema:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- Registrazione di controllo supportata da SQLite per le chiamate allo strumento MCP -- Filtri per strumento, successo/fallimento, chiave API e impaginazione -- Tabella di controllo della dashboard + endpoint statistici per l'automazione
+ - -🔐 21. "Ho bisogno di autorizzazioni MCP con ambito per integrazione" +
+🧰 17. "I need MCP tools as first-class product capabilities" -Client diversi dovrebbero avere accesso con privilegi minimi alle categorie di strumenti. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Come OmniRoute risolve il problema:** +**How OmniRoute solves it:** -- 10 ambiti MCP granulari per l'accesso controllato agli strumenti -- Applicazione dell'ambito e visibilità nell'interfaccia utente di gestione MCP -- Postura predefinita sicura per gli strumenti operativi
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding - -⚙️ 22. "Ho bisogno di controlli operativi senza ridistribuzione" + -I team necessitano di rapidi cambiamenti di runtime durante incidenti o eventi di costo. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**Come OmniRoute risolve il problema:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Cambia l'attivazione combinata direttamente dalla dashboard MCP -- Applicare profili di resilienza da pacchetti di policy predefiniti -- Ripristinare lo stato dell'interruttore dallo stesso pannello operativo
+**How OmniRoute solves it:** - -🔄 23. "Ho bisogno di visibilità e annullamento del ciclo di vita delle attività A2A in tempo reale" +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Senza visibilità del ciclo di vita, gli incidenti relativi alle attività diventano difficili da valutare. + -**Come OmniRoute risolve il problema:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Elenco/filtro delle attività per stato/competenza con impaginazione -- Esamina i metadati, gli eventi e gli artefatti delle attività -- Endpoint di annullamento dell'attività e azione dell'interfaccia utente con conferma
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. - -🌊 24. "Ho bisogno di metriche di flusso attive per il carico A2A" +**How OmniRoute solves it:** -I flussi di lavoro in streaming richiedono informazioni operative sulla concorrenza e sulle connessioni live. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**Come OmniRoute risolve il problema:** + -- Contatori di flussi attivi integrati nello stato A2A -- Timestamp dell'ultima attività e conteggi per stato -- Schede dashboard A2A per il monitoraggio delle operazioni in tempo reale +
+📋 20. "I need auditable MCP tool execution" - -🪪 25. "Ho bisogno del rilevamento degli agenti standard per i clienti" +When tools mutate config or trigger ops actions, teams need forensic traceability. -I client e gli agenti di orchestrazione esterni necessitano di metadati leggibili dal computer per l'onboarding. +**How OmniRoute solves it:** -**Come OmniRoute risolve il problema:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- Scheda agente esposta in "/.well-known/agent.json". -- Capacità e competenze mostrate nell'interfaccia utente di gestione -- L'API di stato A2A include metadati di rilevamento per l'automazione
+ - -🧭 26. "Ho bisogno della rilevabilità del protocollo nell'UX del prodotto" +
+🔐 21. "I need scoped MCP permissions per integration" -Se gli utenti non riescono a scoprire le superfici del protocollo, l'adozione e la qualità del supporto diminuiscono. +Different clients should have least-privilege access to tool categories. -**Come OmniRoute risolve il problema:** +**How OmniRoute solves it:** -- Pagina**Endpoint**consolidata con schede per endpoint Proxy, MCP, A2A e API -- Commuta lo stato del servizio in linea (online/offline) per MCP e A2A -- Collegamenti dalla panoramica alle schede di gestione dedicate
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling - -🧪 27. "Ho bisogno della convalida del protocollo end-to-end con clienti reali" + -I test simulati non sono sufficienti per verificare la compatibilità del protocollo prima del rilascio. +
+⚙️ 22. "I need operational controls without redeploying" -**Come OmniRoute risolve il problema:** +Teams need quick runtime changes during incidents or cost events. -- Suite E2E che avvia l'app e utilizza il trasporto client SDK MCP reale -- Test client A2A per i flussi di rilevamento, invio, streaming, acquisizione e annullamento -- Effettuare un controllo incrociato delle asserzioni con l'audit MCP e le API delle attività A2A
+**How OmniRoute solves it:** - -📡 28. "Ho bisogno di osservabilità unificata su tutte le interfacce" +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -Suddividere l'osservabilità per protocollo crea punti ciechi e un MTTR più lungo. + -**Come OmniRoute risolve il problema:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Dashboard/registri/analisi unificati in un unico prodotto -- Salute + audit + richiesta di telemetria su livelli OpenAI, MCP e A2A -- API operative per stato e automazione
+Without lifecycle visibility, task incidents become hard to triage. - -💼 29. "Ho bisogno di un runtime per proxy + strumenti + orchestrazione dell'agente" +**How OmniRoute solves it:** -L'esecuzione di numerosi servizi separati aumenta i costi operativi e le modalità di guasto. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**Come OmniRoute risolve il problema:** + -- Proxy compatibile con OpenAI, server MCP e server A2A in uno stack -- Autenticazione condivisa, resilienza, archivio dati e osservabilità -- Modello politico coerente su tutte le superfici di interazione +
+🌊 24. "I need active stream metrics for A2A load" - -🚀 30. "Ho bisogno di distribuire flussi di lavoro agenti senza l'espansione incontrollata del codice" +Streaming workflows require operational insight into concurrency and live connections. -I team perdono velocità quando uniscono più servizi e script ad hoc. +**How OmniRoute solves it:** -**Come OmniRoute risolve il problema:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Strategia endpoint unificata per clienti e agenti -- Interfacce utente di gestione del protocollo integrate e percorsi di convalida del fumo -- Fondamenti pronti per la produzione (sicurezza, registrazione, resilienza, backup)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) @@ -609,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Playbook B: stack di codifica a costo zero**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Playbook C: catena di fallback sempre attiva 24 ore su 24, 7 giorni su 7**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -632,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Playbook D: operazioni dell'agente con MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Configura la codifica AI in pochi minuti a**$ 0/mese**. Connect these free accounts and use the built-in**Free Stack**combo. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Passo | Action | Providers Unlocked | +| Step | Action | Providers Unlocked | | ---- | -------------------------------------------------- | ------------------------------------------------------------------ | -| 1| Connect**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**unlimited** | -| 2| Connect**Qoder**(Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... —**unlimited**| -| 3| Connetti**Qwen**(codice dispositivo) | qwen3-coder-plus, qwen3-coder-flash... —**unlimited** | -| 4| Connect**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180K/mo free** | -| 5| `/dashboard/combos` →**Free Stack ($0)**template | Round-robin all free providers automatically | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Point any IDE/CLI to:**`http://localhost:20128/v1` · API Key: `any-string` · Done. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Copertura extra opzionale (anche gratuita):**Chiave API Groq (30 RPM gratuiti), NVIDIA NIM (40 RPM gratuiti, 70+ modelli), Cerebras (1 milione di tok/giorno), chiave API LongCat (50 milioni di token/giorno!), Cloudflare Workers AI (10.000 neuroni/giorno, 50+ modelli).## Avvio Rapido +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Avvio Rapido ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **Utenti pnpm:**esegui `pnpm approve-builds -g` dopo l'installazione per abilitare gli script di build nativi richiesti da `better-sqlite3` e `@swc/core`: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > > ```bash > pnpm install -g omniroute -> pnpm approve-builds -g # Seleziona tutti i pacchetti → approva -> percorso omnicomprensivo +> pnpm approve-builds -g # Select all packages → approve +> omniroute > ``` -La dashboard si apre in "http://localhost:20128" e l'URL di base dell'API è "http://localhost:20128/v1". +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Comando | Descrizione | -| ----------------------- | ------------------------------------------------------------------ | -| `omnipercorso` | Avvia il server (`PORT=20128`, API e dashboard sulla stessa porta) | -| `omniroute --port 3000` | Imposta la porta canonica/API su 3000 | -| `omniroute --mcp` | Avvia il server MCP (trasporto stdio) | -| `omniroute --no-open` | Non aprire automaticamente il browser | -| `omniroute --help` | Mostra aiuto | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Modalità porta divisa opzionale:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -Per la maggior parte delle distribuzioni sono necessari solo: +For most deployments, you only need: -| Variabile | Predefinito | Scopo | -| ------------------------ | ----------------------- | --------------------------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600000` | Base di riferimento condivisa per recupero upstream, timeout Undici nascosti, richieste di impronte digitali TLS e timeout proxy/richieste bridge API | -| `STREAM_IDLE_TIMEOUT_MS` | eredita `REQUEST_TIMEOUT_MS` | Intervallo massimo tra i blocchi di streaming prima che OmniRoute interrompa il flusso SSE | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -La compatibilità con le versioni precedenti è preservata: `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` esistenti e altre variabili di timeout per livello continuano a funzionare e sovrascrivono la linea di base condivisa. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Se hai bisogno di un controllo più preciso, sono disponibili sostituzioni avanzate:| Variabile | Predefinito | Scopo | -| --------------------------------------- | ----------------------------------- | -------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | eredita `REQUEST_TIMEOUT_MS` | Timeout totale della richiesta upstream utilizzato dal segnale di interruzione del recupero principale | -| `FETCH_HEADERS_TIMEOUT_MS` | eredita `FETCH_TIMEOUT_MS` | Limite temporale Undici per la ricezione delle intestazioni di risposta upstream | -| `FETCH_BODY_TIMEOUT_MS` | eredita `FETCH_TIMEOUT_MS` | Limite di tempo Undici tra i blocchi del corpo upstream (`0` lo disabilita) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Timeout connessione TCP Undici | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | "4000" | Timeout del socket keep-alive inattivo Undici | -| "TLS_CLIENT_TIMEOUT_MS" | eredita `FETCH_TIMEOUT_MS` | Timeout per le richieste di impronte digitali TLS effettuate tramite `wreq-js` | -| "API_BRIDGE_PROXY_TIMEOUT_MS" | eredita `REQUEST_TIMEOUT_MS` o `30000` | Timeout per l'inoltro proxy `/v1` dalla porta API alla porta del dashboard | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | "max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)" | Timeout della richiesta in entrata sul server bridge API | -| "API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS" | `60000` | Timeout dell'intestazione in entrata sul server bridge API | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Timeout keep-alive sul server bridge API | -| "API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS" | "0" | Timeout di inattività del socket sul server bridge API (`0` lo disabilita) | +Advanced overrides are available if you need finer control: -Se esegui OmniRoute dietro Nginx, Caddy, Cloudflare o un altro proxy inverso, assicurati che il proxy -i timeout sono anche superiori ai timeout di flusso/recupero di OmniRoute.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Apri Dashboard → "Provider" e connetti almeno un fornitore (OAuth o chiave API). -2. Apri Dashboard → "Endpoint" e crea una chiave API. -3. (Facoltativo) Apri Dashboard → "Combo" e imposta la catena di fallback.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Funziona con Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode e SDK compatibili con OpenAI.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (per operazioni guidate da strumenti):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -Quindi collega il tuo client MCP su "stdio" e testa strumenti come: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (per flussi di lavoro da agente ad agente):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -771,13 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` - -Void Linux (modello `xbps-src`) +
+Void Linux (`xbps-src` template) -Per gli utenti Void Linux, è possibile creare un pacchetto nativo utilizzando "xbps-src". Salva questo blocco come `srcpkgs/omniroute/template`:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -789,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -797,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -873,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -884,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute è disponibile come immagine Docker pubblica su [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Corsa veloce:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -894,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**Con file di ambiente:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Utilizzo di Docker Compose:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Il supporto della dashboard per le distribuzioni Docker ora include un**Cloudflare Quick Tunnel**con un solo clic su "Dashboard → Endpoint". La prima abilitazione scarica `cloudflared` solo quando necessario, avvia un tunnel temporaneo verso il tuo attuale endpoint `/v1` e mostra l'URL `https://*.trycloudflare.com/v1` generato direttamente sotto il normale URL pubblico. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Note: +Notes: -- Gli URL del tunnel rapido sono temporanei e cambiano dopo ogni riavvio. -- I tunnel rapidi non vengono ripristinati automaticamente dopo il riavvio di OmniRoute o del contenitore. Riattivarli dalla dashboard quando necessario. -- L'installazione gestita attualmente supporta Linux, macOS e Windows su "x64"/"arm64". -- I tunnel rapidi gestiti utilizzano per impostazione predefinita il trasporto HTTP/2 per evitare avvisi rumorosi del buffer QUIC UDP in ambienti container vincolati. Imposta `CLOUDFLARED_PROTOCOL=quic` o `auto` se desideri un trasporto diverso. -- Le immagini Docker raggruppano le root CA del sistema e le passano a "cloudflared" gestito, evitando errori di attendibilità TLS quando il tunnel si avvia all'interno del contenitore. -- SQLite funziona in modalità WAL. È necessario consentire il completamento di "docker stop" in modo che OmniRoute possa verificare le ultime modifiche in "storage.sqlite". -- I file Compose in bundle impostano già un periodo di tolleranza di 40 secondi. Se esegui l'immagine direttamente, mantieni `--stop-timeout 40` (o simile) in modo che gli arresti manuali non interrompano la pulizia dello spegnimento. -- Imposta `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` se desideri che OmniRoute utilizzi un file binario esistente invece di scaricarne uno. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Utilizzo di Docker Compose con Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute può essere esposto in modo sicuro utilizzando il provisioning SSL automatico di Caddy. Assicurati che il record DNS A del tuo dominio punti all'IP del tuo server.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` - -| Immagine | Etichetta | Taglia | Descrizione | +| Image | Tag | Size | Description | | ------------------------ | -------- | ------ | --------------------- | -| `diegosouzapw/omniroute` | `ultimo` | ~250 MB | Ultima versione stabile | -| `diegosouzapw/omniroute` | `1.0.3` | ~250 MB | Versione attuale |--- +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | + +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**NOVITÀ!**OmniRoute è ora disponibile come**applicazione desktop nativa**per Windows, macOS e Linux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Esegui OmniRoute come app desktop autonoma: nessun terminale, nessun browser, nessuna connessione Internet richiesta per i modelli locali. L'app basata su Electron include: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Finestra nativa**: finestra dell'app dedicata con integrazione nella barra delle applicazioni -- 🔄**Avvio automatico**: avvia OmniRoute all'accesso al sistema -- 🔔**Notifiche native**: ricevi avvisi in caso di esaurimento della quota o problemi con il provider -- ⚡**Installazione con un clic**: NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Modalità offline**: funziona completamente offline con il server in bundle### Avvio Rapido +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Avvio Rapido ```bash # Development mode @@ -983,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -Quando ridotto a icona, OmniRoute si trova nella barra delle applicazioni con azioni rapide: +When minimized, OmniRoute lives in your system tray with quick actions: -- Apri il cruscotto -- Cambia la porta del server -- Esci dall'applicazione +- Open dashboard +- Change server port +- Quit application -📖 Documentazione completa: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Livello | Fornitore | Costo | Reimpostazione quota | Ideale per | -| ------------------ | --------------------------- | ------------------------------------- | ------------------------ | ------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 ABBONAMENTO** | Codice Claude (Pro) | $20/mese | 5 ore + settimanale | Già iscritto | -| | Codice (Plus/Pro) | $20-200/mese | 5 ore + settimanale | Utenti OpenAI | -| | Gemelli CLI | **GRATIS** | 180K/mese + 1K/giorno | Tutti! | -| | Copilota GitHub | $ 10-19/mese | Mensile | Utenti GitHub | -| **🔑 CHIAVE API** | NVIDIA NIM | **GRATUITO**(sviluppatore per sempre) | ~40 giri al minuto | Oltre 70 modelli aperti | -| | Cerebri | **GRATUITO**(1 milione di tok/giorno) | 60.000 TPM/30 giri/min | Il più veloce del mondo | -| | Groq | **GRATIS**(30 GIRI) | 14,4K RPD | Lama/Gemma ultraveloce | -| | DeepSeek V3.2 | $ 0,27/$ 1,10 per 1 milione | Nessuno | Miglior ragionamento prezzo/qualità | -| | xAI Grok-4 Veloce | **$ 0,20/$ 0,50 per 1 milione**🆕 | Nessuno | Più veloce + chiamata strumento, ultrabassa | -| | xAI Grok-4 (standard) | $ 0,20/$ 1,50 per 1 milione 🆕 | Nessuno | Fiore all'occhiello del ragionamento di xAI | -| | Maestrale | Prova gratuita + pagamento | Tariffa limitata | IA europea | -| | OpenRouter | Pagamento in base all'uso | Nessuno | Oltre 100 modelli aggr. | -| **💰 ECONOMICO** | GLM-5 (via Z.AI) 🆕 | $ 0,5/1 milione | Tutti i giorni 10:00 | Uscita 128K, nuova ammiraglia | -| | GLM-4.7 | $ 0,6/1 milione | Tutti i giorni 10:00 | Backup del budget | -| | MiniMax M2.5 🆕 | Ingresso di $ 0,3/1 milione | 5 ore di rotazione | Ragionamento + compiti agentici | -| | MiniMax M2.1 | $ 0,2/1 milione | 5 ore di rotazione | Opzione più economica | -| | Kimi K2.5 (API Moonshot) 🆕 | Pagamento in base all'uso | Nessuno | Accesso diretto all'API Moonshot | -| | Kimi K2 | $ 9/mese fisso | 10 milioni di token/mese | Costo prevedibile | -| **🆓 GRATUITO** | Qoder | **$0** | Illimitato | 5 modelli illimitati | -| | Qwen | **$0** | Illimitato | 4 modelli illimitati | -| | Kiro | **$0** | Illimitato | Claude Sonetto/Haiku (costruttore AWS) | -| | LongCat Flash-Lite 🆕 | **$0**(50 milioni di tok/giorno 🔥) | 1 RPS | La più grande quota gratuita sulla Terra | -| | Impollinazioni AI 🆕 | **$0**(nessuna chiave necessaria) | 1 richiesta/15s | GPT-5, Claude, DeepSeek, Lama 4 | -| | Cloudflare Workers AI 🆕 | **$0**(10.000 neuroni/giorno) | ~150 risposte/giorno | Oltre 50 modelli, vantaggio globale | -| | Scaleway AI 🆕 | **$0**(totale di 1 milione di token) | Tariffa limitata | UE/GDPR, Qwen3 235B, Llama 70B | > 🆕**Nuovi modelli aggiunti (marzo 2026):**Famiglia Grok-4 Fast a $ 0,20/$ 0,50/milione (benchmark a 1143 ms — 30% più veloce di Gemini 2.5 Flash), GLM-5 tramite Z.AI con output 128K, ragionamento MiniMax M2.5, prezzo aggiornato DeepSeek V3.2, Kimi K2.5 tramite Moonshot Direct API. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 Stack combinato da $ 0: la configurazione gratuita completa:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Costo zero. Non smette mai di scrivere codice.**Configuralo come una combinazione OmniRoute e tutti i fallback verranno eseguiti automaticamente, senza alcun passaggio manuale.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Tutti i modelli riportati di seguito sono**gratuiti al 100% e non è richiesta alcuna carta di credito**. OmniRoute esegue automaticamente i percorsi tra di loro quando una quota si esaurisce: combinali tutti per una combinazione indistruttibile a $ 0.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Modello | Prefisso | Limite | Limite di velocità | +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | --------------------- | -| `claude-sonetto-4.5` | `kr/` |**Illimitato**| Nessun limite giornaliero riportato | -| `claude-haiku-4.5` | `kr/` |**Illimitato**| Nessun limite giornaliero segnalato | -| `claude-opus-4.6` | `kr/` |**Illimitato**| Ultima opera tramite Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -| Modello | Prefisso | Limite | Limite di velocità | +### 🟢 QODER MODELS (Free PAT via qodercli) + +| Model | Prefix | Limit | Rate Limit | | ------------------ | ------ | ------------- | --------------- | -| `kimi-k2-pensiero` | `se/` |**Illimitato**| Nessun limite riportato | -| `qwen3-coder-plus` | `se/` |**Illimitato**| Nessun limite riportato | -| `deepseek-r1` | `se/` |**Illimitato**| Nessun limite segnalato | -| `minimax-m2.1` | `se/` |**Illimitato**| Nessun limite riportato | -| `kimi-k2` | `se/` |**Illimitato**| Nessun limite segnalato | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -> Metodo di connessione consigliato:**Token di accesso personale + `qodercli`**. OAuth del browser lo è -> sperimentale e disabilitato per impostazione predefinita a meno che non siano configurate le variabili di ambiente `QODER_OAUTH_*`.### 🟡 QWEN MODELS (Device Code Auth) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| Modello | Prefisso | Limite | Limite di velocità | +### 🟡 QWEN MODELS (Device Code Auth) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | ------------------- | -| `qwen3-coder-plus` | `qw/` |**Illimitato**| Nessun limite riportato | -| `qwen3-coder-flash` | `qw/` |**Illimitato**| Nessun limite riportato | -| `qwen3-coder-next` | `qw/` |**Illimitato**| Nessun limite riportato | -| `modello-visione` | `qw/` |**Illimitato**| Multimodale (immagini) |### 🟣 GEMINI CLI (Google OAuth) +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Modello | Prefisso | Limite | Limite di velocità | -| ------------------------ | ------ | --------------------- | ------------- | -| `gemini-3-flash-anteprima` | `gc/` |**180K tok/mese**+ 1K/giorno | Reset mensile | -| `gemini-2.5-pro` | `gc/` | 180K/mese (pool condiviso) | Alta qualità |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +### 🟣 GEMINI CLI (Google OAuth) -| Livello | Limite giornaliero | Limite di velocità | Note | +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | + +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---------- | ------------ | ----------- | ------------------------------------------------------ | -| Libero (Sviluppo) | Nessun limite massimo |**~40 giri/min**| Oltre 70 modelli; transizione ai limiti tariffari puri a metà del 2025 | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -Modelli gratuiti popolari: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -| Livello | Limite giornaliero | Limite di velocità | Note | -| ---- | ----------------- | ---------------- | -------------------------------------------------- | -| Gratuito |**1 milione di token/giorno**| 60.000 TPM/30 giri/min | L'inferenza LLM più veloce al mondo; si ripristina quotidianamente | +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -Disponibili gratuitamente: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -| Livello | Limite giornaliero | Limite di velocità | Note | +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` + +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---- | ------------- | ---------------- | ----------------------------------------- | -| Gratuito |**14,4K RPD**| 30 giri/min per modello | Nessuna carta di credito; 429 al limite, non addebitato | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -Disponibili gratuitamente: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -| Modello | Prefisso | Quota gratuita giornaliera | Note | -| ----------------------- | ------ | ----------------- | ----------------------- | -| `LongCat-Flash-Lite` | `lc/` |**50 milioni di token**💥 | La quota gratuita più grande di sempre | -| `LongCat-Flash-Chat` | `lc/` | Gettoni da 500.000 | Chat multigiro | -| `LongCat-Flash-Thinking` | `lc/` | Gettoni da 500.000 | Ragionamento/CoT | -| `LongCat-Flash-Thinking-2601` | `lc/` | Gettoni da 500.000 | Versione gennaio 2026 | -| `LongCat-Flash-Omni-2603` | `lc/` | Gettoni da 500.000 | Multimodale | +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 -> 100% gratuito durante la beta pubblica. Iscriviti a [longcat.chat](https://longcat.chat) tramite e-mail o telefono. Si ripristina ogni giorno alle 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | -| Modello | Prefisso | Limite di velocità | Fornitore dietro | +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. + +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| `openai` | `pol/` | 1 richiesta/15s | GPT-5 | -| `claude` | `pol/` | 1 richiesta/15s | Claude antropico | -| `gemelli` | `pol/` | 1 richiesta/15s | Google Gemelli | -| `ricerca profonda` | `pol/` | 1 richiesta/15s | DeepSeek V3 | -| `lama` | `pol/` | 1 richiesta/15s | Meta Lama 4 Esploratore | -| `maestrale` | `pol/` | 1 richiesta/15s | Maestrale AI | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**Zero attrito:**Nessuna registrazione, nessuna chiave API. Aggiungi il provider Pollinations con un campo chiave vuoto e funzionerà immediatamente.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| Livello | Neuroni giornalieri | Utilizzo equivalente | Note | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 + +| Tier | Daily Neurons | Equivalent Usage | Notes | | ---- | ------------- | --------------------------------------- | ----------------------- | -| Gratuito |**10.000**| ~150 risposte LLM / 500 audio / 15.000 incorporamenti | Vantaggio globale, oltre 50 modelli | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -Modelli gratuiti popolari: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (audio gratuito!), `@cf/qwen/qwen2.5-coder-15b-instruct` +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -> Richiede token API + ID account da [dash.cloudflare.com](https://dash.cloudflare.com). Memorizza l'ID account nelle impostazioni del provider.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -| Livello | Quota libera | Posizione | Note | +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | | ---- | ------------- | ------------ | ----------------------------------- | -| Gratuito |**1 milione di gettoni**| 🇫🇷 Parigi, UE | Nessuna carta di credito necessaria entro i limiti | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | -Disponibile gratuitamente: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` -> Conforme all'UE/GDPR. Ottieni la chiave API su [console.scaleway.com](https://console.scaleway.com). +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). ->**💡 Lo stack gratuito definitivo (11 fornitori, $ 0 per sempre):** +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Sonnet/Haiku ILLIMITATO -> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 ILLIMITATO -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 milioni di token al giorno 🔥 -> Impollinazioni (pol/) → GPT-5, Claude, DeepSeek, Llama 4: nessuna chiave necessaria -> Qwen (qw/) → modelli qwen3-coder ILLIMITATI -> Gemini (gemini/) → Gemini 2.5 Flash — 1.500 richieste/giorno gratis -> Cloudflare AI (cf/) → Oltre 50 modelli: 10.000 neuroni al giorno -> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1 milione di token gratuiti (UE) -> Groq (groq/) → Lama/Gemma — 14.4K richieste/giorno ultraveloci -> NVIDIA NIM (nvidia/) → Oltre 70 modelli aperti: 40 RPM per sempre -> Cerebras (cerebras/) → Lama/Qwen il più veloce al mondo — 1 milione di tok/giorno -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Trascrivi qualsiasi audio/video per**$ 0**: Deepgram guida con $ 200 gratuiti, AssemblyAI $ 50 di riserva, Groq Whisper come backup di emergenza illimitato. +## 🎙️ Free Transcription Combo -| Fornitore | Crediti gratuiti | Miglior modello | Limite di velocità | -| ----------------- | ---------------------- | -------------------------------------------- | ---------------------- | -| 🟢**Deepgram**|**$200 gratuiti**(iscrizione) | `nova-3`: massima precisione, oltre 30 lingue | Nessun limite RPM sui crediti gratuiti | -| 🔵**AssembleaAI**|**$50 gratuiti**(iscrizione) | `universal-3-pro`: capitoli, sentimento, PII | Nessun limite RPM sui crediti gratuiti | -| 🔴**Groq**|**Gratis per sempre**| `sussurro-large-v3` — OpenAI Whisper | 30 giri/min (velocità limitata) | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. -**Combinazione suggerita in `/dashboard/combos`:**``` +| Provider | Free Credits | Best Model | Rate Limit | +| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | + +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Quindi in `/dashboard/media` → scheda**Trascrizione**: carica qualsiasi file audio o video → seleziona il tuo endpoint combinato → ottieni la trascrizione nei formati supportati.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 è costruito come piattaforma operativa, non solo come proxy di inoltro.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Caratteristica | Cosa fa | -| ------------------------------------------ | -------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Grok-4 Fast Family** | Modelli xAI a 0,20 $/0,50 $/milione: 1.143 ms con benchmark (30% più veloce di Gemini 2.5 Flash) | -| 🧠**GLM-5 via Z.AI** | Contesto di output da 128.000, $ 0,5/1 milione: il nuovo fiore all'occhiello della famiglia GLM | -| 🔮**MiniMax M2.5** | Ragionamento + compiti di agente a 0,30 dollari/1 milione: miglioramento significativo rispetto a M2.1 | -| 🎯**toolCalling Flag per modello** | `toolCalling: true/false` per modello nel registro: AutoCombo ignora i modelli non compatibili con lo strumento | -| 🌍**Rilevamento dell'intento multilingue** | Parole chiave PT/ZH/ES/AR nel punteggio AutoCombo: migliore selezione del modello per contenuti non inglesi | -| 📊**Falback basati sul benchmark** | La latenza p95 reale dalle richieste in tempo reale alimenta il punteggio combinato: AutoCombo apprende dai dati effettivi | -| 🔁**Richiedi deduplicazione** | Finestra di deduplicazione basata sull'hash dei contenuti: sicura per più agenti, impedisce addebiti duplicati | -| 🔌**Strategia router collegabile** | Interfaccia `RouterStrategy` estensibile: aggiungi logica di routing personalizzata come plug-in | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Caratteristica | Cosa fa | -| ---------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Parco giochi modello** | Pagina dashboard per testare direttamente qualsiasi modello: selettori provider/modello/endpoint, Monaco Editor, streaming, interruzione, tempistica | -| 🔏**Corrispondenza impronta digitale CLI** | Ordinamento dell'intestazione/corpo per provider in modo che corrisponda alle firme CLI native: attiva/disattiva per provider in Impostazioni > Sicurezza.**Il tuo IP proxy viene preservato** | -| 🤝**Supporto ACP (protocollo agente client)** | Rilevamento agente CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw + altri 9), generatore di processi, endpoint `/api/acp/agents` | -| 🤖**Dashboard agenti ACP** | Debug › Pagina Agenti: griglia di 14 agenti con stato di installazione, versione, modulo agente personalizzato per qualsiasi strumento CLI. Gli utenti**OpenCode**ricevono un pulsante "Scarica opencode.json" che genera automaticamente una configurazione pronta per l'uso con tutti i modelli disponibili. | -| 🔧**Instradamento del modello personalizzato `apiFormat`** | I modelli personalizzati con `apiFormat: "responses"` ora vengono indirizzati correttamente al traduttore dell'API Responses | -| 🏢**Codice Isolamento dello spazio di lavoro** | Aree di lavoro Codex multiple per e-mail: OAuth separa correttamente le connessioni in base all'ID dell'area di lavoro | -| 🔄**Aggiornamento automatico Electron** | L'app desktop verifica la disponibilità di aggiornamenti + installazione automatica al riavvio | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Caratteristica | Cosa fa | -| --------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**Server MCP (25 strumenti)** | Strumenti IDE/agente tramite 3 trasporti: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memoria + 4 strumenti di abilità | -| 🤝**Server A2A (JSON-RPC + SSE)** | Esecuzione di attività da agente ad agente con flussi di sincronizzazione e streaming | -| 🧭**Pagina Endpoint consolidati** | Pagina di gestione a schede con schede Endpoint Proxy, MCP, A2A e Endpoint API | -| 🎚️**Abilita/Disabilita servizio** | Interruttori ON/OFF per MCP e A2A con persistenza delle impostazioni (default: OFF) | -| 🛰️**MCP Runtime Heartbeat** | Stato reale del processo (pid, tempo di attività, età dell'heartbeat, trasporto, modalità ambito) | -| 📋**Pista di controllo MCP** | Registri di controllo filtrabili con successo/fallimento e attribuzione chiave | -| 🔐**Applicazione dell'ambito MCP** | 10 autorizzazioni di ambito granulare per l'accesso controllato agli strumenti | -| 📡**Gestione del ciclo di vita delle attività A2A** | Elenca/filtra attività, ispeziona eventi/artefatti, annulla attività in esecuzione | -| 📋**Scoperta Carta Agente** | `/.well-known/agent.json` per il rilevamento automatico del client | -| 🧪**Cablaggio di prova protocollo E2E** | Il client Real MCP SDK + A2A scorre in `test:protocols:e2e` | -| ⚙️**Controlli operativi** | Cambia combo, applica profili di resilienza, reimposta gli interruttori da un'unica superficie di controllo | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Caratteristica | Cosa fa | -| ------------------------------------------------ | ----------------------------------------------------------------------------------------------- | ----------------------- | -| 🎯**Ripiego intelligente a 4 livelli** | Percorso automatico: Abbonamento → Chiave API → Economico → Gratuito | -| 📊**Monitoraggio delle quote in tempo reale** | Conteggio dei token in tempo reale + reimpostazione del conto alla rovescia per provider | -| 🔄**Traduzione del formato** | OpenAI ↔ Claude ↔ Gemini ↔ Risposte con conversioni sicure per schema | -| 👥**Supporto per più account** | Conti multipli per fornitore con selezione intelligente | -| 🔄**Aggiornamento automatico token** | I token OAuth si aggiornano automaticamente con il nuovo tentativo | -| 🎨**Combo personalizzati** | 9 strategie di bilanciamento + controllo della catena di fallback | -| 🌐**Router con caratteri jolly** | `provider/*` instradamento dinamico | -| 🧠**Pensare al controllo del budget** | Limiti di ragionamento passthrough, automatico, personalizzato e adattivo | -| 🔀**Alias ​​modello** | Aliasing del modello integrato + personalizzato e sicurezza della migrazione | -| ⚡**Degradazione dello sfondo** | Instradare le attività in background a bassa priorità verso modelli più economici | -| 🧪**Routing intelligente basato sulle attività** | Seleziona automaticamente il modello per tipo di contenuto (codifica/visione/analisi/riepilogo) | -| 🔄**Flussi di lavoro dell'agente A2A** | Orchestratore FSM deterministico per esecuzioni di agenti multi-step con stato | -| 🔀**Percorso adattivo** | Override della strategia dinamica basata sul volume dei token e sulla complessità dei prompt | -| 🎲**Diversità dei fornitori** | Punteggio entropico di Shannon che bilancia la distribuzione del traffico combinato automatico | -| 💬**Iniezione richiesta di sistema** | Controlli del comportamento globale applicati in modo coerente | -| 📄**Compatibilità API risposte** | Supporto completo `/v1/responses` per Codex e flussi di lavoro avanzati con agenti | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Caratteristica | Cosa fa | -| ----------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**Generazione di immagini** | `/v1/images/ generations` con cloud e backend locali | -| 📐**Incorporamenti** | `/v1/embeddings` per le pipeline di ricerca e RAG | -| 🎤**Trascrizione audio** | `/v1/audio/transcriptions` — 7 fornitori (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), rilevamento automatico della lingua, supporto MP4/MP3/WAV | -| 🔊**Sintesi vocale** | `/v1/audio/speech` — 10 fornitori (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) con messaggi di errore corretti | -| 🎬**Generazione video** | `/v1/videos/generazioni` (flussi di lavoro ComfyUI + SD WebUI) | -| 🎵**Generazione musicale** | `/v1/music/ generations` (flussi di lavoro ComfyUI) | -| 🛡️**Moderazioni** | Controlli di sicurezza `/v1/moderazioni` | -| 🔀**Riclassifica** | `/v1/rerank` per il punteggio di pertinenza | -| 🔍**Ricerca sul Web**🆕 | `/v1/search` — 5 provider (Serper, Brave, Perplexity, Exa, Tavily), oltre 6.500 gratuiti al mese, failover automatico, cache | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Caratteristica | Cosa fa | -| ---------------------------------------------------- | ------------------------------------------------------------------------------------------------------------ | -------------------------------- | -| 🔌**Interruttori automatici** | Scatto/recupero per modello con controlli di soglia | -| 🎯**Modelli sensibili agli endpoint** | I modelli personalizzati dichiarano gli endpoint supportati + il formato API | -| 🛡️**Mandria Antituonante** | Mutex + protezioni semaforo su eventi ripetizione/velocità | -| 🧠**Semantica + Cache delle firme** | Riduzione costi/latenza con due livelli di cache | -| ⚡**Richiesta Idempotenza** | Finestra di protezione duplicata | -| 🔒**Spoofing delle impronte digitali TLS** | Impronta digitale TLS simile a un browser:**riduce il rilevamento dei bot e la segnalazione degli account** | -| 🔏**Corrispondenza impronta digitale CLI** | Corrisponde alle firme delle richieste CLI native:**riduce il rischio di ban preservando l'IP proxy** | -| 🌐**Filtro IP** | Controllo della lista consentita/lista bloccata per le distribuzioni esposte | -| 📊**Limiti di velocità modificabili** | Limiti globali/a livello di provider configurabili con persistenza | -| 📉**Degradazione graziosa** | Fallback con funzionalità multilivello che proteggono le operazioni principali del gateway | -| 📜**Traccia di controllo della configurazione** | Tracciamento delle modifiche basato sulle differenze che impedisce la deriva operativa con semplici rollback | -| ⏳**Sincronizzazione integrità fornitore** | Monitoraggio proattivo della scadenza dei token che attiva avvisi prima degli errori di autorizzazione | -| 🚪**Disabilita automaticamente gli account esclusi** | Interruttore automatico che sigilla automaticamente gli account token bloccati in modo permanente | -| 🔑**Gestione delle chiavi API e ambito** | Emissione/rotazione sicura delle chiavi e controlli del modello/fornitore | -| 👁️**Rivelazione chiave API con ambito**🆕 | Attiva il ripristino delle chiavi API tramite `ALLOW_API_KEY_REVEAL` | -| 🛡️**Protetti `/models`** | Autenticazione opzionale e nascondiglio del provider per il catalogo dei modelli | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Caratteristica | Cosa fa | -| ----------------------------------------- | ------------------------------------------------------------------------------------ | ---------------------------- | -| 📝**Richiesta + Registrazione proxy** | Richiesta/risposta completa e registrazione proxy | -| 📉**Registri dettagliati in streaming**🆕 | Ricostruisce in modo pulito i flussi di payload SSE nell'interfaccia utente | -| 📋**Dashboard dei registri unificati** | Visualizzazioni richieste, proxy, audit e console in un'unica pagina | -| 🔍**Richiedi telemetria** | latenza p50/p95/p99 e tracciamento delle richieste | -| 🏥**Dashboard della salute** | Tempo di attività, stati degli interruttori, blocchi, statistiche della cache | -| 💰**Monitoraggio dei costi** | Budget controls and per-model pricing visibility | -| 📈**Visualizzazioni analitiche** | Approfondimenti sull'utilizzo del modello/fornitore e visualizzazioni delle tendenze | -| 🧪**Quadro di valutazione** | Test del set d'oro con strategie di corrispondenza configurabili | -| 📡**Diagnostica in tempo reale**🆕 | Bypass semantico della cache per test live combinati accurati | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Caratteristica | Cosa fa | -| -------------------------------------------- | -------------------------------------------------------------------------------- | --------------------- | -| 🌐**Distribuisci ovunque** | Localhost, VPS, Docker, ambienti cloud | -| 🚇**Tunnel Cloudflare**🆕 | Integrazione Quick Tunnel con un clic dalla dashboard | -| 🔑**Filtro modello chiave API** | Risposta nativa /v1/models filtrata tramite i ruoli di contesto Bearer assegnati | -| ⚡**Bypass intelligente della cache** | Euristica TTL configurabile e controlli di recupero forzato | -| 🔄**Backup/Ripristino** | Flussi di import/export e disaster recovery | -| 🧙**Procedura guidata di inserimento** | Configurazione guidata al primo avvio | -| 🔧**Dashboard degli strumenti CLI** | Configurazione con un clic per gli strumenti di codifica più diffusi | -| 🎮**Parco giochi modello** | Testa qualsiasi provider/modello/endpoint dalla dashboard | -| 🔏**Attiva/disattiva impronta digitale CLI** | Corrispondenza dell'impronta digitale per provider in Impostazioni > Sicurezza | -| 🌐**i18n (30 lingue)** | Full dashboard + docs language support with RTL coverage | -| 🧹**Cancella tutti i modelli** | Cancellazione dell'elenco dei modelli con un clic nei dettagli del fornitore | -| 👁️**Controlli della barra laterale**🆕 | Nascondi componenti e integrazioni da Impostazioni aspetto | -| 📋**Modelli di problemi** | Modelli GitHub standardizzati per bug e funzionalità | -| 📂**Directory dati personalizzata** | Sostituzione di `DATA_DIR` per la posizione di archiviazione | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1296,103 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -Quando la quota, la tariffa o l'integrità vengono meno, OmniRoute passa automaticamente al candidato successivo senza passaggio manuale.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A sono rilevabili nell'interfaccia utente e nei documenti (non nascosti) -- Le API di stato del protocollo espongono dati operativi in tempo reale (`/api/mcp/*`, `/api/a2a/*`) -- I dashboard includono azioni per le operazioni del secondo giorno (attivazioni/disattivazione combo, ripristino degli interruttori, annullamento delle attività)#### Translator + validation workflow +#### Protocol management that is visible and operable -L'area Traduttore comprende: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Parco giochi**: richiedi controlli di trasformazione -**Chat Tester**: richiesta/risposta completa andata e ritorno -**Banco di prova**: più casi in un'unica esecuzione -**Live Monitor**: visualizzazione del traffico in tempo reale +#### Translator + validation workflow -Inoltre convalida del protocollo con client reali tramite `npm run test:protocols:e2e`. +The Translator area includes: -> 📖**[README del server MCP](open-sse/mcp-server/README.md)**: riferimenti allo strumento, configurazioni IDE ed esempi di client +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[README del server A2A](src/lib/a2a/README.md)**: competenze, metodi JSON-RPC, streaming e ciclo di vita delle attività## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute include un framework di valutazione integrato per testare la qualità della risposta LLM rispetto a un golden set. Access it via**Analytics → Evals**in the dashboard.### Built-in Golden Set +## 🧪 Evaluations (Evals) -L'"OmniRoute Golden Set" precaricato contiene casi di test per: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Saluti, matematica, geografia, generazione di codici -- Conformità al formato JSON, traduzione, generazione di markdown -- Rifiuto di sicurezza (contenuto dannoso), conteggio, logica booleana### Evaluation Strategies +### Built-in Golden Set -| Strategia | Descrizione | Esempio | -| ---------------- | -------------------------------------------------------------------------------------- | ----------------------------------- | --- | -| `esatto` | L'output deve corrispondere esattamente a | `"4"` | -| `contains` | L'output deve contenere una sottostringa (senza distinzione tra maiuscole e minuscole) | `"Parigi"` | -| `regex` | L'output deve corrispondere al modello regex | `"1.*2.*3"` | -| "personalizzato" | La funzione JS personalizzata restituisce vero/falso | `(output) => output.lunghezza > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) - -🧩 Configurazione MCP (Model Context Protocol) +
+🧩 MCP Setup (Model Context Protocol) -Avvia il trasporto MCP in modalità stdio:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Flusso di convalida consigliato: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Connetti il tuo client MCP su stdio. -2. Esegui `omniroute_get_health`. -3. Esegui `omniroute_list_combos`. -4. Apri "/dashboard/mcp" per confermare battito cardiaco, attività e controllo. +Useful APIs for automation: -API utili per l'automazione: +- `GET /api/mcp/status` +- `GET /api/mcp/tools` +- `GET /api/mcp/audit` +- `GET /api/mcp/audit/stats` -- "OTTIENI /api/mcp/status". -- `OTTIENI /api/mcp/tools` -- "OTTIENI /api/mcp/audit". -- "OTTIENI /api/mcp/audit/stats".
+ - -🤝 Configurazione A2A (Agent2Agent) +
+🤝 A2A Setup (Agent2Agent) -Scopri l'agente:```bash +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Invia un'attività:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` +Manage lifecycle: -Gestisci il ciclo di vita: - -- "OTTIENI /api/a2a/status". -- "OTTIENI /api/a2a/tasks". +- `GET /api/a2a/status` +- `GET /api/a2a/tasks` - `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -Interfaccia utente operativa: +Operational UI: -- "/dashboard/a2a" per l'osservabilità di attività/stato/flusso e azioni di fumo
+- `/dashboard/a2a` for task/state/stream observability and smoke actions - -🧪 Convalida del protocollo end-to-end + -Convalida entrambi i protocolli con client reali:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` This verifies: -- Connessione/elenco/chiamata del client SDK MCP -- Rilevamento/invio/streaming/acquisizione/annullamento A2A -- Controllo incrociato dei dati nell'audit MCP e nelle API di gestione delle attività A2A
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs - -💳 Fornitori di abbonamenti### Claude Code (Pro/Max) + + +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1405,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Suggerimento professionale:**usa Opus per attività complesse, Sonnet per la velocità. OmniRoute tiene traccia della quota per modello!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1419,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Ogni account Codex ora dispone di opzioni di attivazione/disattivazione delle policy in `Dashboard -> Provider`: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- "5h" (ON/OFF): applica la politica di soglia della finestra di 5 ore. -- "Settimanale" (ON/OFF): applica la politica di soglia della finestra settimanale. -- Comportamento soglia: quando una finestra abilitata raggiunge un utilizzo >=90%, quell'account viene saltato. -- Comportamento di rotazione: OmniRoute instrada automaticamente al successivo account Codex idoneo. -- Comportamento di ripristino: allo scadere del tempo "resetAt" del provider, l'account diventa nuovamente idoneo automaticamente. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Scenari: +Scenarios: -- `5h ON` + `Weekly ON`: l'account viene saltato quando una delle finestre raggiunge la soglia. -- `5h OFF` + `Weekly ON`: solo l'utilizzo settimanale può bloccare l'account. -- `5h ON` + `Weekly OFF`: solo un utilizzo di 5 ore può bloccare l'account. -- `resetAt` superato: l'account rientra automaticamente nella rotazione (nessuna riattivazione manuale).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1444,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Miglior rapporto qualità-prezzo:**Enorme livello gratuito! Utilizzalo prima dei livelli a pagamento.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1459,71 +1662,91 @@ Models:
- -🔑 Provider di chiavi API### NVIDIA NIM (FREE developer access — 70+ models) +
+🔑 API Key Providers -1. Iscriviti: [build.nvidia.com](https://build.nvidia.com) -2. Ottieni la chiave API gratuita (1000 crediti di inferenza inclusi) -3. Dashboard → Aggiungi fornitore → NVIDIA NIM: - - Chiave API: `nvapi-your-key` +### NVIDIA NIM (FREE developer access — 70+ models) -**Modelli:**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` e oltre 50 altri +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Suggerimento professionale:**API compatibile con OpenAI: funziona perfettamente con la traduzione del formato di OmniRoute!### DeepSeek +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -1. Iscriviti: [platform.deepseek.com](https://platform.deepseek.com) -2. Ottieni la chiave API -3. Dashboard → Aggiungi fornitore → DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -**Modelli:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +### DeepSeek -1. Iscriviti: [console.groq.com](https://console.groq.com) -2. Ottieni la chiave API (livello gratuito incluso) -3. Dashboard → Aggiungi fornitore → Groq +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -**Modelli:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Suggerimento da professionista:**Inferenza ultraveloce: ideale per la codifica in tempo reale!### OpenRouter (100+ Models) +### Groq (Free Tier Available!) -1. Iscriviti: [openrouter.ai](https://openrouter.ai) -2. Ottieni la chiave API -3. Dashboard → Aggiungi provider → OpenRouter +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -**Modelli:**accedi a oltre 100 modelli di tutti i principali fornitori tramite un'unica chiave API. +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Comportamento della dashboard:**i modelli OpenRouter sono gestiti da**Modelli disponibili**. L'aggiunta manuale, l'importazione e la sincronizzazione automatica aggiornano tutti lo stesso elenco.
+**Pro Tip:** Ultra-fast inference — best for real-time coding! - -💰 Fornitori economici (backup)### GLM-4.7 (Daily reset, $0.6/1M) +### OpenRouter (100+ Models) -1. Iscriviti: [Zhipu AI](https://open.bigmodel.cn/) -2. Ottieni la chiave API dal piano di codifica -3. Dashboard → Aggiungi chiave API: - - Fornitore: `glm` - - Chiave API: "la tua chiave". +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -**Utilizzare:**`glm/glm-4.7` +**Models:** Access 100+ models from all major providers through a single API key. -**Suggerimento professionale:**Il piano di codifica offre una quota 3× a un costo di 1/7! Resetta ogni giorno alle 10:00.### MiniMax M2.1 (5h reset, $0.20/1M) +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -1. Iscriviti: [MiniMax](https://www.minimax.io/) -2. Ottieni la chiave API -3. Dashboard → Aggiungi chiave API + -**Utilizzo:**`minimax/MiniMax-M2.1` +
+💰 Cheap Providers (Backup) -**Suggerimento professionale:**Opzione più economica per contesti lunghi (token da 1 milione)!### Kimi K2 ($9/month flat) +### GLM-4.7 (Daily reset, $0.6/1M) -1. Iscriviti: [Moonshot AI](https://platform.moonshot.ai/) -2. Ottieni la chiave API -3. Dashboard → Aggiungi chiave API +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**Utilizzare:**`kimi/kimi-latest` +**Use:** `glm/glm-4.7` -**Suggerimento da professionista:**Risolti 9$ al mese per 10 milioni di token = 0,90$/1 milione di costi effettivi!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. - -🆓 Provider GRATUITI (backup di emergenza)### Qoder (5 FREE models via OAuth) +### MiniMax M2.1 (5h reset, $0.20/1M) + +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `minimax/MiniMax-M2.1` + +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1564,8 +1787,10 @@ Models:
- -🎨Crea combo### Example 1: Maximize Subscription → Cheap Backup +
+🎨 Create Combos + +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1593,8 +1818,10 @@ Cost: $0 forever!
- -🔧Integrazione CLI### Cursor IDE +
+🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1605,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Utilizza la pagina**Strumenti CLI**nel dashboard per la configurazione con un clic o modifica manualmente `~/.claude/settings.json`.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1616,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Opzione 1: Dashboard (consigliata):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Opzione 2 — Manuale:**Modifica `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1633,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Nota:**OpenClaw funziona solo con OmniRoute locale. Utilizza "127.0.0.1" invece di "localhost" per evitare problemi di risoluzione IPv6.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1647,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Passaggio 1:**aggiungi OmniRoute come provider personalizzato:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Passaggio 2:**Crea/modifica `opencode.json` nella root del tuo progetto:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1673,117 +1909,130 @@ opencode } } } -```` +``` -**Passaggio 3:**Seleziona il modello in OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Suggerimento:**aggiungi qualsiasi modello disponibile nel tuo endpoint OmniRoute `/v1/models` alla sezione `models`. Utilizza il formato "provider/id-modello" dal dashboard OmniRoute.
+ --- ## Risoluzione dei Problemi - -Fai clic per espandere la guida alla risoluzione dei problemi +
+Click to expand troubleshooting guide -**"Il modello linguistico non ha fornito messaggi"** +**"Language model did not provide messages"** - Provider quota exhausted → Check dashboard quota tracker -- Soluzione: utilizzare il fallback combinato o passare al livello più economico +- Solution: Use combo fallback or switch to cheaper tier -**Limitazione della velocità** +**Rate limiting** -- Quota di abbonamento esaurita → Fallback su GLM/MiniMax -- Aggiungi combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**Token OAuth scaduto** +**OAuth token expired** -- Aggiornato automaticamente da OmniRoute -- Se i problemi persistono: Dashboard → Provider → Riconnetti +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Costi elevati** +**High costs** -- Controlla le statistiche di utilizzo in Dashboard → Costi -- Passa dal modello principale a GLM/MiniMax -- Utilizza il livello gratuito (Gemini CLI, Qoder) per attività non critiche +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Le porte dashboard/API sono sbagliate** +**Dashboard/API ports are wrong** -- "PORT" è la porta base canonica (e la porta API per impostazione predefinita) -- "API_PORT" sovrascrive solo il listener API compatibile con OpenAI -- `DASHBOARD_PORT` sovrascrive solo la dashboard/il listener Next.js -- Imposta "NEXT_PUBLIC_BASE_URL" sul tuo dashboard/URL pubblico (per callback OAuth) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Errori di sincronizzazione cloud** +**Cloud sync errors** -- Verifica che `BASE_URL` punti alla tua istanza in esecuzione -- Verifica che "CLOUD_URL" punti all'endpoint cloud previsto -- Mantieni i valori `NEXT_PUBLIC_*` allineati con i valori lato server +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Primo accesso non funzionante** +**First login not working** -- Controlla "INITIAL_PASSWORD" in ".env". -- Se non impostata, la password di fallback è "123456". +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Nessun registro delle richieste** +**No request logs** -- Gli elementi della richiesta vengono scritti in "DATA_DIR/call_logs/" come un file JSON per richiesta -- Abilita l'acquisizione della pipeline da Dashboard → Log → Richiedi log se hai bisogno di payload dettagliati per fase -- Imposta `APP_LOG_TO_FILE=true` se desideri anche i log della console dell'applicazione in `logs/application/app.log` -- Modifica `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` e `CALL_LOG_MAX_ENTRIES` secondo necessità +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Il test di connessione mostra "Non valido" per i provider compatibili con OpenAI** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Molti provider non espongono un endpoint `/models` -- OmniRoute v1.0.6+ include la convalida di fallback tramite completamenti di chat -- Assicurati che l'URL di base includa il suffisso "/v1".### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ Importante per gli utenti che utilizzano OmniRoute su un VPS, Docker o qualsiasi server remoto**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -I provider**Antigravity**e**Gemini CLI**utilizzano**Google OAuth 2.0**. Google richiede che "redirect_uri" nel flusso OAuth corrisponda esattamente a uno degli URI preregistrati nella Google Cloud Console dell'app. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -Le credenziali OAuth incluse in OmniRoute sono registrate**solo per `localhost`**. Quando accedi a OmniRoute su un server remoto (ad esempio `https://omniroute.myserver.com`), Google rifiuta l'autenticazione con:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Devi creare un**ID client OAuth 2.0**in Google Cloud Console con l'URI del tuo server.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Apri Google Cloud Console** +#### Step-by-step -Vai a: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Crea un nuovo ID client OAuth 2.0** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Fai clic su**"+ Crea credenziali"**→**"ID client OAuth"** -- Tipo di applicazione:**"Applicazione Web"** -- Nome: qualsiasi cosa tu voglia (ad esempio "OmniRoute Remote") +**2. Create a new OAuth 2.0 Client ID** -**3. Aggiungi URI di reindirizzamento autorizzati** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -Nel campo**"URI di reindirizzamento autorizzati"**, aggiungi:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Sostituisci `your-server.com` con il dominio o l'IP del tuo server (includi la porta se necessario, ad esempio `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. Salva e copia le credenziali** +After creating, Google will show the **Client ID** and **Client Secret**. -Dopo la creazione, Google mostrerà l'**ID cliente**e il**Segreto cliente**. +**5. Set environment variables** -**5. Imposta variabili di ambiente** +In your `.env` (or Docker environment variables): -Nel tuo `.env` (o variabili di ambiente Docker):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1792,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Riavvia OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Prova a connetterti di nuovo** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Dashboard → Provider → Antigravity (o Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google ora reindirizzerà correttamente a "https://your-server.com/callback".--- +--- #### Temporary workaround (without custom credentials) -Se non desideri impostare le tue credenziali adesso, puoi comunque utilizzare il**flusso URL manuale**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute apre l'URL di autorizzazione di Google -2. Dopo l'autorizzazione, Google tenta di reindirizzare a "localhost" (che fallisce sul server remoto) -3.**Copia l'URL completo**dalla barra degli indirizzi del browser (anche se la pagina non viene caricata) -4. Incolla l'URL nel campo mostrato nella modalità di connessione OmniRoute -5. Fai clic su**"Connetti"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Funziona perché il codice di autorizzazione nell'URL è valido indipendentemente dal fatto che la pagina di reindirizzamento sia stata caricata.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. - -🇧🇷 Versione in portoghese#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -I fornitori**Antigravity**e**Gemini CLI**utilizzano**Google OAuth 2.0**per l'autenticazione. Google richiede che `redirect_uri` utilizzato nel flusso OAuth sia**esattamente**uno degli URI predefiniti nell'applicazione Google Cloud Console. +
+🇧🇷 Versão em Português -Le credenziali OAuth supportate su OmniRoute sono cadastrada**solo per `localhost`**. Quando accedi a OmniRoute su un server remoto (es: `https://omniroute.meuservidor.com`), Google rifiuta l'autenticazione come:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -È necessario creare un**OAuth 2.0 Client ID**su Google Cloud Console come URI del proprio server.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Accesso a Google Cloud Console** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -**2. Crea un nuovo ID client OAuth 2.0** +**2. Crie um novo OAuth 2.0 Client ID** -- Fare clic su**"+ Crea credenziali"**→**"ID client OAuth"** -- Tipo di applicazione:**"Applicazione Web"** -- Nome: escolha qualquer nome (es: `OmniRoute Remote`) +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Aggiunta come URI di reindirizzamento autorizzati** +**3. Adicione as Authorized Redirect URIs** -Nessun campo**"URI di reindirizzamento autorizzati"**, aggiunta:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Sostituisci `seu-servidor.com` con il tuo dominio o IP con il tuo server (inclusa la porta se necessaria, ad esempio: `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. Salva e copia come credenziale** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Dopo aver creato, Google mostrerà il**Client ID**e il**Client Secret**. +**5. Configure as variáveis de ambiente** -**5. Configura come variáveis de ambiente** +No seu `.env` (ou nas variáveis de ambiente do Docker): -No seu `.env` (o nelle varie impostazioni dell'ambiente Docker):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1871,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Riavvio di OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute +``` -```` +**7. Tente conectar novamente** -**7. Tenete collegato novamente** +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Dashboard → Provider → Antigravity (o Gemini CLI) → OAuth +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. -Ora Google verrà reindirizzato correttamente a `https://seu-servidor.com/callback` e l'autenticazione funzionerà.--- +--- #### Workaround temporário (sem configurar credenciais próprias) -Se non vuoi creare credenziali proprie adesso, puoi anche usare il flusso**manuale dell'URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. OmniRoute aprirà l'URL di autorizzazione di Google -2. Dopo aver autorizzato Google tenterà di reindirizzare a "localhost" (che non funziona sul server remoto) -3.**Copiare l'URL completo**dalla barra degli indirizzi del browser (anche se la pagina non viene caricata) -4. Inserire questo URL nel campo visualizzato nella modalità di connessione di OmniRoute -5. Fare clic su**"Connetti"** +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) +4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute +5. Clique em **"Connect"** -> Questa soluzione alternativa funziona perché il codice di autorizzazione sull'URL è valido indipendentemente dal reindirizzamento caricato o meno.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1909,64 +2171,73 @@ Se non vuoi creare credenziali proprie adesso, puoi anche usare il flusso**manua ## 🛠️ Tech Stack - -Fai clic per espandere i dettagli dello stack tecnologico +
+Click to expand tech stack details --**Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ è**non supportato**— i file binari nativi `better-sqlite3` sono incompatibili) --**Lingua**: TypeScript 5.9 —**100% TypeScript**su `src/` e `open-sse/` (zero `any` nei moduli principali dalla v2.0) --**Framework**: Next.js 16 + React 19 + Tailwind CSS 4 --**Database**: LowDB (JSON) + SQLite (stato del dominio + log proxy + audit MCP + decisioni di routing) --**Schemi**: Zod (convalida I/O dello strumento MCP, contratti API) --**Protocolli**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Streaming**: eventi inviati dal server (SSE) --**Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization --**Test**: test runner Node.js + Vitest (oltre 900 test inclusi unità, integrazione, E2E) --**CI/CD**: Azioni GitHub (pubblicazione npm automatica + Docker Hub al rilascio) --**Sito web**: [omniroute.online](https://omniroute.online) --**Pacchetto**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Resilienza**: interruttore automatico, backoff esponenziale, mandria anti-tuono, spoofing TLS, autoriparazione automatica combo
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Documentazione -| Documento | Descrizione | -| ----------------------------------------------------- | --------------------------------------------------- | -| [Guida per l'utente](docs/USER_GUIDE.md) | Provider, combinazioni, integrazione CLI, distribuzione | -| [Riferimento API](docs/API_REFERENCE.md) | Tutti gli endpoint con esempi | -| [Server MCP](open-sse/mcp-server/README.md) | 16 strumenti MCP, configurazioni IDE, client Python/TS/Go | -| [Server A2A](src/lib/a2a/README.md) | Protocollo JSON-RPC 2.0, competenze, streaming, gestione delle attività | -| [Motore Auto-Combo](docs/auto-combo.md) | Punteggio a 6 fattori, pacchetti modalità, autoriparazione | -| [Risoluzione dei problemi](docs/TROUBLESHOOTING.md) | Problemi comuni e soluzioni | -| [Architettura](docs/ARCHITECTURE.md) | Architettura del sistema e componenti interni | -| [Contribuire](CONTRIBUTING.md) | Impostazione e linee guida per lo sviluppo | -| [Specifiche OpenAPI](docs/openapi.yaml) | Specifica OpenAPI 3.0 | -| [Politica di sicurezza](SECURITY.md) | Segnalazione delle vulnerabilità e pratiche di sicurezza | -| [Distribuzione VM](docs/VM_DEPLOYMENT_GUIDE.md) | Guida completa: configurazione VM + nginx + Cloudflare | -| [Galleria delle funzionalità](docs/FEATURES.md) | Tour visivo della dashboard con screenshot | -| [Elenco di controllo del rilascio](docs/RELEASE_CHECKLIST.md) | Passaggi di convalida prima del rilascio |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute ha**oltre 210 funzionalità pianificate**in più fasi di sviluppo. Ecco le aree chiave: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Categoria | Planned Features | In evidenza | -| ----------------------- | ---------------- | -------------------------------------------------------------------------------------- | -| 🧠**Routing e intelligenza**| 25+ | Routing a latenza più bassa, routing basato su tag, verifica preliminare delle quote, selezione dell'account P2C | -| 🔒**Sicurezza e conformità**| 20+ | Rafforzamento SSRF, cloaking delle credenziali, limite di velocità per endpoint, ambito delle chiavi di gestione | -| 📊**Osservabilità**| 15+ | Integrazione OpenTelemetry, monitoraggio delle quote in tempo reale, monitoraggio dei costi per modello | -| 🔄**Integrazioni del provider**| 20+ | Registro dei modelli dinamici, tempi di recupero dei provider, Codex multi-account, analisi delle quote Copilot | -| ⚡**Prestazioni**| 15+ | Doppio livello di cache, cache dei prompt, cache delle risposte, streaming keepalive, API batch | -| 🌐**Ecosistema**| 10+ | API WebSocket, ricarica a caldo della configurazione, archivio di configurazione distribuito, modalità commerciale |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**Integrazione OpenCode**: supporto nativo del provider per l'IDE di codifica AI OpenCode -- 🔗**Integrazione TRAE**: supporto completo per il framework di sviluppo AI TRAE -- 📦**API Batch**: elaborazione batch asincrona per richieste collettive -- 🎯**Routing basato su tag**: instrada le richieste in base a tag e metadati personalizzati -- 💰**Strategia a costo più basso**: seleziona automaticamente il fornitore più economico disponibile +### 🔜 Coming Soon -> 📝 Specifiche complete delle funzionalità disponibili in [`docs/new-features/`](docs/new-features/) (217 specifiche dettagliate)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1974,18 +2245,20 @@ OmniRoute ha**oltre 210 funzionalità pianificate**in più fasi di sviluppo. Ecc ### How to Contribute -1. Effettuare il fork del repository -2. Crea il ramo della tua funzionalità ("git checkout -b feature/amazing-feature") -3. Applica le tue modifiche (`git commit -m 'Aggiungi funzionalità straordinarie'`) -4. Spingi sul ramo ("git push origin feature/amazing-feature") -5. Aprire una richiesta di pull +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Consulta [CONTRIBUTING.md](CONTRIBUTING.md) per linee guida dettagliate.### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1997,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Un ringraziamento speciale a**[9router](https://github.com/decolua/9router)**di**[decolua](https://github.com/decolua)**— il progetto originale che ha ispirato questo fork. OmniRoute si basa su queste incredibili fondamenta con funzionalità aggiuntive, API multimodali e una riscrittura completa di TypeScript. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Un ringraziamento speciale a**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**: l'implementazione originale di Go che ha ispirato questo port di JavaScript.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Licenza -Licenza MIT: per i dettagli vedere [LICENZA](LICENZA).--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/it/docs/ARCHITECTURE.md b/docs/i18n/it/docs/ARCHITECTURE.md index 9d2a30ca18..53cfb21642 100644 --- a/docs/i18n/it/docs/ARCHITECTURE.md +++ b/docs/i18n/it/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Ultimo aggiornamento: 28-03-2026_## Executive Summary -OmniRoute è un gateway di routing AI locale e un dashboard basato su Next.js. -Fornisce un singolo endpoint compatibile con OpenAI (`/v1/*`) e instrada il traffico attraverso più provider upstream con traduzione, fallback, aggiornamento dei token e monitoraggio dell'utilizzo. -Funzionalità principali: +_Last updated: 2026-03-28_ -- Superficie API compatibile con OpenAI per CLI/strumenti (28 provider) -- Traduzione di richieste/risposte tra formati di fornitori -- Fallback combo modello (sequenza multi-modello) -- Fallback a livello di account (più account per fornitore) -- Gestione della connessione del provider OAuth + chiave API -- Generazione di incorporamenti tramite `/v1/embeddings` (6 fornitori, 9 modelli) -- Generazione di immagini tramite `/v1/images/ generations` (4 fornitori, 9 modelli) -- Analisi dei tag Think (`...`) per modelli di ragionamento -- Sanificazione della risposta per una rigorosa compatibilità con l'SDK OpenAI -- Normalizzazione dei ruoli (sviluppatore→sistema, sistema→utente) per compatibilità tra provider -- Conversione dell'output strutturato (json_schema → Gemini ResponseSchema) -- Persistenza locale per provider, chiavi, alias, combo, impostazioni, prezzi -- Monitoraggio dell'utilizzo/costo e registrazione delle richieste -- Sincronizzazione cloud opzionale per la sincronizzazione multi-dispositivo/stato -- Lista consentita/lista bloccata IP per il controllo dell'accesso API -- Gestione intelligente del budget (passthrough/automatico/personalizzato/adattivo) -- Iniezione rapida del sistema globale -- Monitoraggio della sessione e rilevamento delle impronte digitali -- Limitazione tariffaria migliorata per account con profili specifici del fornitore -- Modello di interruttore automatico per la resilienza del fornitore -- Protezione gregge antituono con bloccaggio mutex -- Cache di deduplicazione delle richieste basata su firma -- Livello dominio: disponibilità del modello, regole di costo, politica di fallback, politica di blocco -- Persistenza dello stato del dominio (cache write-through SQLite per fallback, budget, blocchi, interruttori automatici) -- Motore di policy per la valutazione centralizzata delle richieste (blocco → budget → fallback) -- Richiedi telemetria con aggregazione della latenza p50/p95/p99 -- ID di correlazione (X-Request-Id) per la traccia end-to-end -- Registrazione del controllo di conformità con rinuncia per chiave API -- Quadro di valutazione per la garanzia della qualità LLM -- Dashboard dell'interfaccia utente di resilienza con stato dell'interruttore automatico in tempo reale -- Provider OAuth modulari (12 moduli individuali in `src/lib/oauth/provviders/`) +## Executive Summary -Modello runtime primario: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- I percorsi dell'app Next.js in "src/app/api/\*" implementano sia le API del dashboard che le API di compatibilità -- Un core SSE/routing condiviso in `src/sse/*` + `open-sse/*` gestisce l'esecuzione, la traduzione, lo streaming, il fallback e l'utilizzo del provider## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Runtime del gateway locale -- API di gestione della dashboard -- Autenticazione del provider e aggiornamento del token -- Richiedi traduzione e streaming SSE -- Stato locale + persistenza dell'utilizzo -- Orchestrazione opzionale della sincronizzazione cloud### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Implementazione del servizio cloud dietro `NEXT_PUBLIC_CLOUD_URL` -- SLA/piano di controllo del fornitore esterno al processo locale -- Gli stessi binari CLI esterni (Claude CLI, Codex CLI, ecc.)## Dashboard Surface (Current) +### Out of Scope -Pagine principali in `src/app/(dashboard)/dashboard/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard`: avvio rapido + panoramica del provider -- `/dashboard/endpoint`: proxy endpoint + MCP + A2A + schede endpoint API -- `/dashboard/providers`: connessioni e credenziali del provider -- `/dashboard/combos`: strategie combinate, modelli, regole di routing del modello -- "/dashboard/costs": aggregazione dei costi e visibilità dei prezzi -- "/dashboard/analytics" — analisi e valutazioni sull'utilizzo -- "/dashboard/limits": controlli di quote/tariffe -- `/dashboard/cli-tools`: onboarding della CLI, rilevamento del runtime, generazione della configurazione -- `/dashboard/agents`: agenti ACP rilevati + registrazione dell'agente personalizzato -- `/dashboard/media`: area giochi per immagini/video/musica -- `/dashboard/search-tools`: test e cronologia del provider di ricerca -- "/dashboard/health": tempo di attività, interruttori automatici, limiti di velocità -- `/dashboard/logs`: registri di richieste/proxy/audit/console -- `/dashboard/settings`: schede delle impostazioni di sistema (generale, routing, impostazioni predefinite combinate, ecc.) -- `/dashboard/api-manager`: ciclo di vita della chiave API e autorizzazioni del modello## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Directory principali: +Main directories: -- `src/app/api/v1/*` e `src/app/api/v1beta/*` per le API di compatibilità -- `src/app/api/*` per le API di gestione/configurazione -- Successivamente riscrive la mappa `next.config.mjs` da `/v1/*` a `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Percorsi di compatibilità importanti: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts`: include modelli personalizzati con `custom: true` -- `src/app/api/v1/embeddings/route.ts` — generazione di incorporamenti (6 provider) -- `src/app/api/v1/images/ generations/route.ts` — generazione di immagini (4+ fornitori incluso Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — chat dedicata per provider -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — incorporamenti dedicati per provider -- `src/app/api/v1/providers/[provider]/images/ generations/route.ts`: immagini dedicate per provider +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` -- `src/app/api/v1beta/models/[...percorso]/route.ts` +- `src/app/api/v1beta/models/[...path]/route.ts` -Domini di gestione: +Management domains: -- Autenticazione/impostazioni: `src/app/api/auth/*`, `src/app/api/settings/*` -- Provider/connessioni: `src/app/api/provviders*` -- Nodi del provider: `src/app/api/provider-nodes*` -- Modelli personalizzati: `src/app/api/provider-models` (GET/POST/DELETE) -- Catalogo modelli: `src/app/api/models/route.ts` (GET) -- Configurazione proxy: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- Chiavi/alias/combo/prezzi: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- Utilizzo: `src/app/api/usage/*` -- Sincronizzazione/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` -- Supporti per gli strumenti CLI: `src/app/api/cli-tools/*` -- Filtro IP: `src/app/api/settings/ip-filter` (GET/PUT) -- Budget pensante: `src/app/api/settings/thinking-budget` (GET/PUT) -- Prompt di sistema: `src/app/api/settings/system-prompt` (GET/PUT) -- Sessioni: `src/app/api/sessions` (GET) -- Limiti di velocità: `src/app/api/rate-limits` (GET) -- Resilienza: "src/app/api/resilience" (GET/PATCH): profili del fornitore, interruttore automatico, stato limite di velocità -- Ripristino della resilienza: `src/app/api/resilience/reset` (POST): ripristina gli interruttori + tempi di recupero -- Statistiche della cache: `src/app/api/cache/stats` (GET/DELETE) -- Disponibilità del modello: `src/app/api/models/availability` (GET/POST) -- Telemetria: `src/app/api/telemetry/summary` (GET) +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) - Budget: `src/app/api/usage/budget` (GET/POST) -- Catene di fallback: `src/app/api/fallback/chains` (GET/POST/DELETE) -- Controllo di conformità: `src/app/api/compliance/audit-log` (GET) -- Valutazioni: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- Politiche: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -Principali moduli di flusso: +## 2) SSE + Translation Core -- Voce: `src/sse/handlers/chat.ts` -- Orchestrazione principale: `open-sse/handlers/chatCore.ts` -- Adattatori di esecuzione del provider: `open-sse/executors/*` -- Rilevamento formato/configurazione provider: `open-sse/services/provider.ts` -- Analisi/risoluzione del modello: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Logica di fallback dell'account: `open-sse/services/accountFallback.ts` -- Registro di traduzione: `open-sse/translator/index.ts` -- Trasformazioni del flusso: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- Estrazione/normalizzazione dell'utilizzo: `open-sse/utils/usageTracking.ts` +Main flow modules: + +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` - Think tag parser: `open-sse/utils/thinkTagParser.ts` -- Gestore di incorporamento: `open-sse/handlers/embeddings.ts` -- Incorporamento del registro del provider: `open-sse/config/embeddingRegistry.ts` -- Gestore di generazione di immagini: `open-sse/handlers/imageGeneration.ts` -- Registro del fornitore di immagini: `open-sse/config/imageRegistry.ts` -- Sanificazione della risposta: `open-sse/handlers/responseSanitizer.ts` -- Normalizzazione del ruolo: `open-sse/services/roleNormalizer.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -Servizi (logica aziendale): +Services (business logic): -- Selezione/punteggio dell'account: `open-sse/services/accountSelector.ts` -- Gestione del ciclo di vita del contesto: `open-sse/services/contextManager.ts` -- Applicazione del filtro IP: `open-sse/services/ipFilter.ts` -- Tracciamento della sessione: `open-sse/services/sessionManager.ts` -- Richiedi la deduplicazione: `open-sse/services/signatureCache.ts` -- Inserimento del prompt del sistema: `open-sse/services/systemPrompt.ts` -- Gestione intelligente del budget: `open-sse/services/thinkingBudget.ts` -- Routing del modello jolly: `open-sse/services/wildcardRouter.ts` -- Gestione dei limiti di tariffa: `open-sse/services/rateLimitManager.ts` -- Interruttore automatico: `open-sse/services/circuitBreaker.ts` +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -Moduli del livello di dominio: +Domain layer modules: -- Disponibilità del modello: `src/lib/domain/modelAvailability.ts` -- Regole/budget di costo: `src/lib/domain/costRules.ts` -- Politica di fallback: `src/lib/domain/fallbackPolicy.ts` -- Risolutore combinato: `src/lib/domain/comboResolver.ts` -- Politica di blocco: `src/lib/domain/lockoutPolicy.ts` -- Motore delle politiche: `src/domain/policyEngine.ts` — blocco centralizzato → budget → valutazione fallback -- Catalogo dei codici di errore: `src/lib/domain/errorCodes.ts` -- ID richiesta: `src/lib/domain/requestId.ts` -- Timeout di recupero: `src/lib/domain/fetchTimeout.ts` -- Richiedi telemetria: `src/lib/domain/requestTelemetry.ts` -- Conformità/controllo: `src/lib/domain/compliance/index.ts` -- Corridore di valutazione: `src/lib/domain/evalRunner.ts` -- Persistenza dello stato del dominio: `src/lib/db/domainState.ts` — SQLite CRUD per catene di fallback, budget, cronologia dei costi, stato di blocco, interruttori automatici +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -Moduli provider OAuth (12 file singoli in `src/lib/oauth/provviders/`): +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -- Indice del registro: `src/lib/oauth/provviders/index.ts` -- Singoli fornitori: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` -- Thin wrapper: `src/lib/oauth/provviders.ts` — riesporta da singoli moduli## 3) Persistence Layer +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -DB di stato primario (SQLite): +## 3) Persistence Layer -- Infrastruttura core: `src/lib/db/core.ts` (better-sqlite3, migrazioni, WAL) -- Riesportazione della facciata: `src/lib/localDb.ts` (livello di compatibilità sottile per i chiamanti) -- file: `${DATA_DIR}/storage.sqlite` (o `$XDG_CONFIG_HOME/omniroute/storage.sqlite` se impostato, altrimenti `~/.omniroute/storage.sqlite`) -- entità (tabelle + spazi dei nomi KV): providerConnections, providerNodes, modelAliases, combos, apiKeys, impostazioni, prezzi,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +Primary state DB (SQLite): -Persistenza dell'utilizzo: +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -- facciata: `src/lib/usageDb.ts` (moduli scomposti in `src/lib/usage/*`) -- Tabelle SQLite in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` -- rimangono elementi di file opzionali per compatibilità/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- I file JSON legacy vengono migrati su SQLite dalle migrazioni di avvio quando presenti +Usage persistence: -DB dello stato del dominio (SQLite): +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- `src/lib/db/domainState.ts` — Operazioni CRUD per lo stato del dominio -- Tabelle (create in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- Schema cache write-through: le mappe in memoria sono autorevoli in fase di esecuzione; le mutazioni vengono scritte in modo sincrono su SQLite; lo stato viene ripristinato dal DB all'avvio a freddo## 4) Auth + Security Surfaces +Domain State DB (SQLite): -- Autenticazione cookie dashboard: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- Generazione/verifica della chiave API: `src/shared/utils/apiKey.ts` -- I segreti del provider sono persistenti nelle voci "providerConnections". -- Supporto proxy in uscita tramite `open-sse/utils/proxyFetch.ts` (env vars) e `open-sse/utils/networkProxy.ts` (configurabile per provider o globale)## 5) Cloud Sync +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start -- Programmazione init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Attività periodica: `src/shared/services/cloudSyncScheduler.ts` -- Attività periodica: `src/shared/services/modelSyncScheduler.ts` -- Controlla il percorso: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Le decisioni di fallback sono guidate da "open-sse/services/accountFallback.ts" utilizzando codici di stato ed euristica dei messaggi di errore. Il routing combinato aggiunge un'ulteriore protezione: gli errori 400 con ambito provider, come gli errori di blocco del contenuto upstream e di convalida del ruolo, vengono trattati come errori locali del modello in modo che le destinazioni combinate successive possano ancora essere eseguite.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -L'aggiornamento durante il traffico live viene eseguito all'interno di `open-sse/handlers/chatCore.ts` tramite l'esecutore `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -La sincronizzazione periodica viene attivata da "CloudSyncScheduler" quando il cloud è abilitato.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -File di archiviazione fisica: +Physical storage files: -- DB di runtime primario: `${DATA_DIR}/storage.sqlite` -- richieste di righe di log: `${DATA_DIR}/log.txt` (artefatto compat/debug) -- archivi strutturati del payload delle chiamate: `${DATA_DIR}/call_logs/` -- sessioni di debug traduttore/richiesta opzionali: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: API di compatibilità -- `src/app/api/v1/providers/[provider]/*`: percorsi dedicati per provider (chat, incorporamenti, immagini) -- `src/app/api/providers*`: CRUD del provider, convalida, test -- `src/app/api/provider-nodes*`: gestione personalizzata dei nodi compatibili -- `src/app/api/provider-models`: gestione dei modelli personalizzati (CRUD) -- `src/app/api/models/route.ts`: API del catalogo modelli (alias + modelli personalizzati) -- `src/app/api/oauth/*`: flussi OAuth/codice dispositivo -- `src/app/api/keys*`: ciclo di vita della chiave API locale -- `src/app/api/models/alias`: gestione degli alias -- `src/app/api/combos*`: gestione delle combo fallback -- `src/app/api/pricing`: il prezzo sostituisce il calcolo dei costi -- `src/app/api/settings/proxy`: configurazione del proxy (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: test di connettività proxy in uscita (POST) -- `src/app/api/usage/*`: API di utilizzo e log -- `src/app/api/sync/*` + `src/app/api/cloud/*`: sincronizzazione cloud e helper rivolti al cloud -- `src/app/api/cli-tools/*`: scrittori/controllori di configurazione CLI locali -- `src/app/api/settings/ip-filter`: lista consentita/lista bloccata IP (GET/PUT) -- `src/app/api/settings/thinking-budget`: configurazione del budget del token pensante (GET/PUT) -- `src/app/api/settings/system-prompt`: prompt di sistema globale (GET/PUT) -- `src/app/api/sessions`: elenco delle sessioni attive (GET) -- `src/app/api/rate-limits`: stato del limite di tariffa per account (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: analisi delle richieste, gestione delle combo, ciclo di selezione dell'account -- `open-sse/handlers/chatCore.ts`: traduzione, invio dell'esecutore, gestione dei tentativi/aggiornamenti, impostazione dello streaming -- `open-sse/executors/*`: comportamento di rete e formato specifico del provider### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: registro e orchestrazione dei traduttori -- Richiedi traduttori: `open-sse/translator/request/*` -- Traduttori di risposta: `open-sse/translator/response/*` -- Costanti di formato: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: configurazione/stato persistente e persistenza del dominio su SQLite -- `src/lib/localDb.ts`: riesportazione della compatibilità per i moduli DB -- `src/lib/usageDb.ts`: facciata della cronologia di utilizzo/registri delle chiamate sopra le tabelle SQLite## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Ogni provider dispone di un esecutore specializzato che estende "BaseExecutor" (in "open-sse/executors/base.ts"), che fornisce la creazione di URL, la costruzione di intestazioni, i tentativi con backoff esponenziale, gli hook di aggiornamento delle credenziali e il metodo di orchestrazione "execute()". +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Esecutore testamentario | Fornitore/i | Movimentazione speciale | -| ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------------------- | -| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Configurazione URL/intestazione dinamica per provider | -| "Esecutore Antigravità" | Google Antigravità | ID progetto/sessione personalizzati, analisi Riprova dopo | -| `CodexExecutor` | Codice OpenAI | Inserisce istruzioni di sistema, forza lo sforzo di ragionamento | -| `CursorExecutor` | Cursore IDE | Protocollo ConnectRPC, codifica Protobuf, firma della richiesta tramite checksum | -| "GithubExecutor" | Copilota GitHub | Aggiornamento del token Copilot, intestazioni che imitano VSCode | -| "KiroExecutor" | AWS CodeWhisperer/Kiro | Formato binario AWS EventStream → conversione SSE | -| "GeminiCLIExecutor" | Gemelli CLI | Ciclo di aggiornamento del token OAuth di Google | +### Persistence -Tutti gli altri provider (inclusi i nodi compatibili personalizzati) utilizzano "DefaultExecutor".## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Fornitore | Formato | Aut. | Flusso | Non streaming | Aggiornamento token | API di utilizzo | -| --------------------- | --------------- | ----------------------- | ---------------- | ------------- | ------------------- | ------------------------ | ------------------------------ | -| Claudio | claude | Chiave API/OAuth | ✅ | ✅ | ✅ | ⚠️ Solo amministratore | -| Gemelli | gemelli | Chiave API/OAuth | ✅ | ✅ | ✅ | ⚠️ Console cloud | -| Gemelli CLI | gemelli-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Console cloud | -| Antigravità | antigravità | OAuth | ✅ | ✅ | ✅ | ✅ API quota completa | -| OpenAI | openai | Chiave API | ✅ | ✅ | ❌ | ❌ | -| Codice | risposte-openai | OAuth | ✅ forzato | ❌ | ✅ | ✅ Limiti tariffari | -| Copilota GitHub | openai | OAuth + token copilota | ✅ | ✅ | ✅ | ✅Istantanee delle quote | -| Cursore | cursore | Checksum personalizzato | ✅ | ✅ | ❌ | ❌ | -| Kiro | Kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Limiti di utilizzo | -| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Su richiesta | -| Qoder | openai | OAuth (base) | ✅ | ✅ | ✅ | ⚠️ Su richiesta | -| OpenRouter | openai | Chiave API | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | claude | Chiave API | ✅ | ✅ | ❌ | ❌ | -| Ricerca profonda | openai | Chiave API | ✅ | ✅ | ❌ | ❌ | -| Groq | openai | Chiave API | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | openai | Chiave API | ✅ | ✅ | ❌ | ❌ | -| Maestrale | openai | Chiave API | ✅ | ✅ | ❌ | ❌ | -| Perplessità | openai | Chiave API | ✅ | ✅ | ❌ | ❌ | -| Insieme AI | openai | Chiave API | ✅ | ✅ | ❌ | ❌ | -| Fuochi d'artificio AI | openai | Chiave API | ✅ | ✅ | ❌ | ❌ | -| Cerebri | openai | Chiave API | ✅ | ✅ | ❌ | ❌ | -| Coerenza | openai | Chiave API | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | openai | Chiave API | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -I formati sorgente rilevati includono: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. -- "openai". -- "risposte-openai". -- "claude". -- "gemelli". +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | -I formati di destinazione includono: +All other providers (including custom compatible nodes) use the `DefaultExecutor`. -- Chat/risposte OpenAI -- Claudio -- Busta Gemini/Gemini-CLI/Antigravità +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: + +- `openai` +- `openai-responses` +- `claude` +- `gemini` + +Target formats include: + +- OpenAI chat/Responses +- Claude +- Gemini/Gemini-CLI/Antigravity envelope - Kiro -- Cursore +- Cursor -Le traduzioni utilizzano**OpenAI come formato hub**: tutte le conversioni passano attraverso OpenAI come formato intermedio:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format - -```` +``` Translations are selected dynamically based on source payload shape and provider target format. Additional processing layers in the translation pipeline: --**Sanificazione delle risposte**: rimuove i campi non standard dalle risposte in formato OpenAI (sia in streaming che non in streaming) per garantire la rigorosa conformità dell'SDK --**Role normalization**— Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) --**Think tag extraction**— Parses `...` blocks from content into `reasoning_content` field --**Structured output**— Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema`## Supported API Endpoints +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` -| Endpoint | Formato | Gestore | +## Supported API Endpoints + +| Endpoint | Format | Handler | | -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | -| `POST /v1/chat/completions` | Chatta OpenAI | `src/sse/handlers/chat.ts` | -| `POST /v1/messaggi` | Messaggi di Claude | Same handler (auto-detected) | -| `POST /v1/responses` | Risposte OpenAI | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/embeddings` | Incorporamenti OpenAI | `open-sse/handlers/embeddings.ts` | -| `GET /v1/embeddings` | Elenco dei modelli | Percorso API | -| `POST /v1/images/generations` | Immagini OpenAI | `open-sse/handlers/imageGeneration.ts` | -| `OTTIENI /v1/immagini/generazioni` | Elenco dei modelli | Percorso API | -| `POST /v1/provider/{provider}/chat/completions` | Chatta OpenAI | Dedicated per-provider with model validation | -| `POST /v1/provider/{provider}/embeddings` | Incorporamenti OpenAI | Dedicated per-provider with model validation | -| `POST /v1/providers/{provider}/images/generations` | Immagini OpenAI | Dedicato per provider con convalida del modello | -| `POST /v1/messages/count_tokens` | Conteggio gettoni Claude | Percorso API | -| `GET /v1/models` | Elenco modelli OpenAI | Percorso API (chat + incorporamento + immagine + modelli personalizzati) | -| `GET /api/models/catalog` | Catalogo | Tutti i modelli raggruppati per fornitore + tipo | -| `POST /v1beta/models/*:streamGenerateContent` | Nativo dei Gemelli | Percorso API | -| `OTTIENI/INSERISCI/ELIMINA /api/settings/proxy` | Configurazione proxy | Configurazione proxy di rete | -| `POST /api/settings/proxy/test` | Connettività proxy | Endpoint di test di integrità/connettività proxy | -| `GET/POST/DELETE /api/provider-models` | Modelli di provider | Metadati del modello del provider che supportano i modelli disponibili personalizzati e gestiti |## Bypass Handler +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -Il gestore di bypass (`open-sse/utils/bypassHandler.ts`) intercetta le richieste "usa e getta" note dalla CLI di Claude (ping di riscaldamento, estrazioni di titoli e conteggi di token) e restituisce una**risposta falsa**senza consumare token del provider upstream. Questo viene attivato solo quando "User-Agent" contiene "claude-cli".## Request Logger Pipeline +## Bypass Handler -Il logger delle richieste (`open-sse/utils/requestLogger.ts`) fornisce una pipeline di registrazione del debug in 7 fasi, disabilitata per impostazione predefinita, abilitata tramite `ENABLE_REQUEST_LOGS=true`:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -I file vengono scritti in `/logs//` per ogni sessione di richiesta.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- Tempo di recupero dell'account del provider in caso di errori temporanei/velocità/autenticazione -- fallback dell'account prima di fallire la richiesta -- fallback del modello combinato quando il percorso del modello/provider corrente è esaurito## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- controllo preliminare e aggiornamento con nuovo tentativo per i provider aggiornabili -- Nuovo tentativo 401/403 dopo il tentativo di aggiornamento nel percorso principale## 3) Stream Safety +## 2) Token Expiry -- controller di flusso in grado di riconoscere la disconnessione -- flusso di traduzione con flush di fine flusso e gestione "[DONE]". -- fallback della stima dell'utilizzo quando mancano i metadati di utilizzo del provider## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- Sono emersi errori di sincronizzazione ma il runtime locale continua -- Lo scheduler ha una logica che consente di riprovare, ma l'esecuzione periodica attualmente chiama la sincronizzazione a tentativo singolo per impostazione predefinita## 5) Data Integrity +## 3) Stream Safety -- Migrazioni dello schema SQLite e hook di aggiornamento automatico all'avvio -- JSON legacy → percorso di compatibilità della migrazione SQLite## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Origini della visibilità in runtime: +## 4) Cloud Sync Degradation -- log della console da `src/sse/utils/logger.ts` -- aggregati di utilizzo per richiesta in SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- Acquisizioni dettagliate del payload in quattro fasi in SQLite (`request_detail_logs`) quando "settings.detailed_logs_enabled=true" -- registro testuale dello stato della richiesta in `log.txt` (opzionale/compat) -- log di richiesta/traduzione approfonditi opzionali in "logs/" quando "ENABLE_REQUEST_LOGS=true" -- Endpoint di utilizzo del dashboard (`/api/usage/*`) per il consumo dell'interfaccia utente +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -L'acquisizione dettagliata del payload della richiesta memorizza fino a quattro fasi del payload JSON per chiamata instradata: +## 5) Data Integrity -- richiesta grezza ricevuta dal cliente -- richiesta tradotta effettivamente inviata a monte -- risposta del provider ricostruita come JSON; le risposte in streaming vengono compattate nel riepilogo finale più i metadati del flusso -- risposta del cliente finale restituita da OmniRoute; le risposte in streaming vengono archiviate nello stesso modulo di riepilogo compatto## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- Il segreto JWT ("JWT_SECRET") protegge la verifica/firma dei cookie della sessione del dashboard -- Il bootstrap della password iniziale (`INITIAL_PASSWORD`) deve essere configurato esplicitamente per il provisioning di prima esecuzione -- Il segreto HMAC della chiave API (`API_KEY_SECRET`) protegge il formato della chiave API locale generata -- I segreti del provider (chiavi/token API) vengono mantenuti nel DB locale e devono essere protetti a livello di file system -- Gli endpoint di sincronizzazione cloud si basano sull'autenticazione della chiave API e sulla semantica dell'ID macchina## Environment and Runtime Matrix +## Observability and Operational Signals -Variabili d'ambiente utilizzate attivamente dal codice: +Runtime visibility sources: -- App/autenticazione: `JWT_SECRET`, `INITIAL_PASSWORD` -- Memorizzazione: `DATA_DIR` -- Comportamento del nodo compatibile: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Override opzionale della base di archiviazione (Linux/macOS quando `DATA_DIR` non è impostato): `XDG_CONFIG_HOME` -- Hashing di sicurezza: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Registrazione: `ENABLE_REQUEST_LOGS` -- URL di sincronizzazione/cloud: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- Proxy in uscita: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` e varianti minuscole -- Flag funzionalità SOCKS5: "ENABLE_SOCKS5_PROXY", "NEXT_PUBLIC_ENABLE_SOCKS5_PROXY" -- Supporti piattaforma/runtime (non configurazione specifica dell'app): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` e `localDb` condividono la stessa policy di directory di base (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) con migrazione dei file legacy. -2. `/api/v1/route.ts` delega allo stesso generatore di catalogo unificato utilizzato da `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) per evitare la deriva semantica. -3. Il registro delle richieste scrive intestazioni/corpo completi quando abilitato; considera la directory dei log come sensibile. -4. Il comportamento del cloud dipende dal `NEXT_PUBLIC_BASE_URL` corretto e dalla raggiungibilità dell'endpoint cloud. -5. La directory `open-sse/` è pubblicata come `@omniroute/open-sse`**pacchetto spazio di lavoro npm**. Il codice sorgente lo importa tramite `@omniroute/open-sse/...` (risolto da Next.js `transpilePackages`). I percorsi dei file in questo documento utilizzano ancora il nome della directory "open-sse/" per coerenza. -6. I grafici nel dashboard utilizzano**Recharts**(basati su SVG) per visualizzazioni analitiche accessibili e interattive (grafici a barre sull'utilizzo del modello, tabelle di suddivisione dei fornitori con percentuali di successo). -7. I test E2E utilizzano**Playwright**(`tests/e2e/`), eseguiti tramite `npm run test:e2e`. I test unitari utilizzano il**test runner Node.js**(`tests/unit/`), eseguito tramite `npm run test:unit`. Il codice sorgente in `src/` è**TypeScript**(`.ts`/`.tsx`); lo spazio di lavoro `open-sse/` rimane JavaScript (`.js`). -8. La pagina Impostazioni è organizzata in 5 schede: Sicurezza, Routing (6 strategie globali: riempimento prima, round robin, p2c, casuale, meno utilizzato, ottimizzato in termini di costi), Resilienza (limiti di velocità modificabili, interruttore automatico, policy), AI (budget pensato, prompt di sistema, cache dei prompt), Avanzate (proxy).## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- Compila dal sorgente: `npm run build` -- Costruisci l'immagine Docker: `docker build -t omniroute .` -- Avviare il servizio e verificare: -- "OTTIENI /api/impostazioni". -- "OTTIENI /api/v1/models". -- L'URL di base di destinazione della CLI deve essere "http://:20128/v1" quando "PORT=20128" +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` +- `GET /api/v1/models` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/it/docs/FEATURES.md b/docs/i18n/it/docs/FEATURES.md index 680a4c69ae..8283480a18 100644 --- a/docs/i18n/it/docs/FEATURES.md +++ b/docs/i18n/it/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Guida visiva a ogni sezione del dashboard OmniRoute.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Gestisci le connessioni dei provider AI: provider OAuth (Claude Code, Codex, Gemini CLI), provider di chiavi API (Groq, DeepSeek, OpenRouter) e provider gratuiti (Qoder, Qwen, Kiro). Gli account Kiro includono il monitoraggio del saldo del credito: crediti rimanenti, indennità totale e data di rinnovo visibili in Dashboard → Utilizzo.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Crea combinazioni di modelli di routing con 6 strategie: priorità, ponderata, round robin, casuale, meno utilizzata e ottimizzata in termini di costi. Ciascuna combinazione concatena più modelli con fallback automatico e include modelli rapidi e controlli di disponibilità.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Analisi completa dell'utilizzo con consumo di token, stime dei costi, mappe di calore delle attività, grafici di distribuzione settimanale e suddivisioni per fornitore.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Monitoraggio in tempo reale: tempo di attività, memoria, versione, percentili di latenza (p50/p95/p99), statistiche della cache e stati degli interruttori automatici del provider.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Quattro modalità per il debug delle traduzioni API:**Playground**(convertitore di formato),**Chat Tester**(richieste live),**Test Bench**(test batch) e**Live Monitor**(streaming in tempo reale).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Prova qualsiasi modello direttamente dalla dashboard. Seleziona provider, modello ed endpoint, scrivi richieste con Monaco Editor, trasmetti le risposte in tempo reale, interrompi a metà flusso e visualizza le metriche temporali.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Temi di colore personalizzabili per l'intera dashboard. Scegli tra 7 colori preimpostati (corallo, blu, rosso, verde, viola, arancione, ciano) o crea un tema personalizzato scegliendo qualsiasi colore esadecimale. Supporta la modalità chiaro, scuro e di sistema.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Pannello delle impostazioni completo con schede: +Comprehensive settings panel with tabs: --**Generale**: archiviazione di sistema, gestione del backup (database di esportazione/importazione) -**Aspetto**: selettore tema (scuro/chiaro/sistema), temi colore predefiniti e colori personalizzati, visibilità del registro di integrità, controlli di visibilità degli elementi della barra laterale -**Sicurezza**: protezione endpoint API, blocco provider personalizzato, filtraggio IP, informazioni sulla sessione -**Routing**: alias del modello, degrado delle attività in background -**Resilienza**: persistenza del limite di velocità, ottimizzazione degli interruttori automatici, disattivazione automatica degli account esclusi, monitoraggio della scadenza del provider -**Avanzate**: sostituzione della configurazione, audit trail della configurazione, modalità di degradazione del fallback![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Configurazione con un clic per gli strumenti di codifica AI: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor e Factory Droid. Dispone di applicazione/ripristino automatico della configurazione, profili di connessione e mappatura del modello.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Dashboard per il rilevamento e la gestione degli agenti CLI. Mostra una griglia di 14 agenti integrati (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) con: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Stato installazione**: installato/non trovato con rilevamento della versione -**Badge di protocollo**: stdio, HTTP, ecc. -**Agenti personalizzati**: registra qualsiasi strumento CLI tramite modulo (nome, binario, comando di versione, argomenti di spawn) -**Corrispondenza dell'impronta digitale della CLI**: attiva/disattiva per provider per abbinare le firme delle richieste CLI native, riducendo il rischio di ban e preservando l'IP proxy--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Genera immagini, video e musica dalla dashboard. Supporta OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open e MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Registrazione delle richieste in tempo reale con filtraggio per provider, modello, account e chiave API. Mostra i codici di stato, l'utilizzo del token, la latenza e i dettagli della risposta.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Il tuo endpoint API unificato con suddivisione delle funzionalità: completamenti chat, API di risposta, incorporamenti, generazione di immagini, riclassificazione, trascrizione audio, sintesi vocale, moderazioni e chiavi API registrate. Integrazione Cloudflare Quick Tunnel e supporto proxy cloud per l'accesso remoto.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -Creare, definire l'ambito e revocare le chiavi API. Ciascuna chiave può essere limitata a modelli/provider specifici con accesso completo o autorizzazioni di sola lettura. Gestione visiva delle chiavi con monitoraggio dell'utilizzo.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Tracciamento delle azioni amministrative con filtraggio per tipo di azione, attore, destinazione, indirizzo IP e timestamp. Cronologia completa degli eventi di sicurezza.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -App desktop nativa Electron per Windows, macOS e Linux. Esegui OmniRoute come applicazione autonoma con integrazione nella barra delle applicazioni, supporto offline, aggiornamento automatico e installazione con un clic. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Caratteristiche principali: +Key features: -- Polling sulla disponibilità del server (nessuna schermata vuota all'avvio a freddo) -- Vassoio di sistema con gestione delle porte -- Politica sulla sicurezza dei contenuti -- Blocco a istanza singola -- Aggiornamento automatico al riavvio -- Interfaccia utente condizionata alla piattaforma (semaforo macOS, barra del titolo predefinita Windows/Linux) -- Packaging di build Electron rafforzato: i `node_modules` con collegamento simbolico nel bundle autonomo vengono rilevati e rifiutati prima del packaging, impedendo la dipendenza del runtime dalla macchina di build (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Vedi [`electron/README.md`](../electron/README.md) per la documentazione completa. +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/it/docs/TROUBLESHOOTING.md b/docs/i18n/it/docs/TROUBLESHOOTING.md index 0d558a4b34..9bef2b8c33 100644 --- a/docs/i18n/it/docs/TROUBLESHOOTING.md +++ b/docs/i18n/it/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -Problemi comuni e soluzioni per OmniRoute.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Problema | Soluzione | -| ------------------------------------------ | ----------------------------------------------------------------------------------------------- | --- | -| Primo accesso non funzionante | Imposta `INITIAL_PASSWORD` in `.env` (nessun valore predefinito hardcoded) | -| Il dashboard si apre sulla porta sbagliata | Imposta `PORT=20128` e `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Nessun registro delle richieste in `logs/` | Imposta `ENABLE_REQUEST_LOGS=true` | -| EACCES: permesso negato | Imposta `DATA_DIR=/path/to/writable/dir` per sovrascrivere `~/.omniroute` | -| La strategia di routing non viene salvata | Aggiornamento alla v1.4.11+ (correzione dello schema Zod per la persistenza delle impostazioni) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Causa:**Quota del fornitore esaurita. +**Cause:** Provider quota exhausted. -**Correzione:** +**Fix:** -1. Controlla il monitoraggio delle quote del dashboard -2. Utilizza una combinazione con livelli di fallback -3. Passa al livello più economico/gratuito### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Causa:**Quota di abbonamento esaurita. +### Rate Limiting -**Correzione:** +**Cause:** Subscription quota exhausted. -- Aggiunto fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Utilizza GLM/MiniMax come backup economico### OAuth Token Expired +**Fix:** -OmniRoute aggiorna automaticamente i token. Se i problemi persistono: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Dashboard → Fornitore → Riconnetti -2. Elimina e aggiungi nuovamente la connessione del provider--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Verifica che `BASE_URL` punti alla tua istanza in esecuzione (ad esempio, `http://localhost:20128`) -2. Verifica che `CLOUD_URL` punti al tuo endpoint cloud (ad esempio, `https://omniroute.dev`) -3. Mantieni i valori `NEXT_PUBLIC_*` allineati con i valori lato server### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Sintomo:**"Token imprevisto 'd'..." sull'endpoint cloud per chiamate non in streaming. +### Cloud `stream=false` Returns 500 -**Causa:**l'upstream restituisce il payload SSE mentre il client si aspetta JSON. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Soluzione alternativa:**utilizzare "stream=true" per le chiamate dirette sul cloud. Il runtime locale include il fallback SSE→JSON.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Crea una nuova chiave dalla dashboard locale (`/api/keys`) -2. Eseguire la sincronizzazione cloud: Abilita Cloud → Sincronizza ora -3. Le chiavi vecchie/non sincronizzate possono ancora restituire "401" sul cloud--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Controllare i campi di runtime: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. Per la modalità portatile: utilizza la destinazione dell'immagine `runner-cli` (CLI in bundle) -3. Per la modalità di montaggio host: impostare `CLI_EXTRA_PATHS` e montare la directory bin dell'host come di sola lettura -4. Se `installed=true` e `runnable=false`: il binario è stato trovato ma il controllo dello stato non è riuscito### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Controlla le statistiche di utilizzo in Dashboard → Utilizzo -2. Passare dal modello principale a GLM/MiniMax -3. Utilizza il livello gratuito (Gemini CLI, Qoder) per attività non critiche -4. Imposta i budget dei costi per chiave API: Dashboard → Chiavi API → Budget--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Imposta "ENABLE_REQUEST_LOGS=true" nel tuo file ".env". I log vengono visualizzati nella directory "logs/".### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Stato principale: `${DATA_DIR}/storage.sqlite` (provider, combo, alias, chiavi, impostazioni) -- Utilizzo: tabelle SQLite in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + facoltativo `${DATA_DIR}/log.txt` e `${DATA_DIR}/call_logs/` -- Richiedi log: `/logs/...` (quando `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Quando l'interruttore di un provider è APERTO, le richieste vengono bloccate fino alla scadenza del tempo di recupero. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Correzione:** +**Fix:** -1. Vai su**Dashboard → Impostazioni → Resilienza** -2. Controllare la scheda dell'interruttore del provider interessato -3. Fare clic su**Reimposta tutto**per cancellare tutti gli interruttori o attendere la scadenza del tempo di recupero -4. Verificare che il provider sia effettivamente disponibile prima di reimpostare### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Se un provider entra ripetutamente nello stato OPEN: +### Provider keeps tripping the circuit breaker -1. Selezionare**Dashboard → Salute → Salute del provider**per il modello di errore -2. Vai su**Impostazioni → Resilienza → Profili fornitore**e aumenta la soglia di errore -3. Controlla se il provider ha modificato i limiti API o richiede la riautenticazione -4. Esaminare la telemetria della latenza: un'elevata latenza può causare errori basati sul timeout--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Assicurati di utilizzare il prefisso corretto: `deepgram/nova-3` o `assemblyai/best` -- Verificare che il provider sia connesso in**Dashboard → Provider**### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Controlla i formati audio supportati: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- Verificare che la dimensione del file rientri nei limiti del provider (in genere < 25 MB) -- Controlla la validità della chiave API del fornitore nella scheda del fornitore--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Utilizza**Dashboard → Traduttore**per eseguire il debug dei problemi di traduzione del formato: +Use **Dashboard → Translator** to debug format translation issues: -| Modalità | Quando usarlo | -| ------------------------- | ---------------------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Parco giochi** | Confronta i formati di input/output fianco a fianco: incolla una richiesta non riuscita per vedere come viene tradotta | -| **Tester della chat** | Invia messaggi in tempo reale e controlla l'intero payload di richiesta/risposta, comprese le intestazioni | -| **Banco di prova** | Esegui test batch su combinazioni di formati per scoprire quali traduzioni sono interrotte | -| **Monitoraggio dal vivo** | Guarda il flusso di richieste in tempo reale per individuare problemi di traduzione intermittenti | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**I tag Thinking non vengono visualizzati**: controlla se il fornitore di destinazione supporta il pensiero e l'impostazione del budget per il pensiero -**Chiamate dello strumento eliminate**: alcune traduzioni di formato potrebbero eliminare i campi non supportati; verificare in modalità Parco giochi -**Prompt di sistema mancante**— Claude e Gemini gestiscono i prompt di sistema in modo diverso; controllare l'output della traduzione -**L'SDK restituisce una stringa non elaborata anziché un oggetto**— Risolto nella versione 1.1.0: il sanitizer della risposta ora rimuove i campi non standard (`x_groq`, `usage_breakdown` e così via) che causano errori di convalida di OpenAI SDK Pydantic -**GLM/ERNIE rifiuta il ruolo `system`**— Risolto nella versione 1.1.0: il normalizzatore dei ruoli unisce automaticamente i messaggi di sistema nei messaggi utente per modelli incompatibili -**Ruolo "sviluppatore" non riconosciuto**— Risolto il problema nella versione 1.1.0: convertito automaticamente in "sistema" per fornitori non OpenAI -**`json_schema` non funziona con Gemini**— Risolto nella versione 1.1.0: `response_format` è ora convertito in `responseMimeType` + `responseSchema` di Gemini--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- Il limite di velocità automatico si applica solo ai fornitori di chiavi API (non OAuth/abbonamento) -- Verificare che**Impostazioni → Resilienza → Profili fornitore**abbia il limite di velocità automatico abilitato -- Controlla se il provider restituisce i codici di stato "429" o le intestazioni "Retry-After".### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -I profili dei fornitori supportano queste impostazioni: +### Tuning exponential backoff --**Ritardo base**: tempo di attesa iniziale dopo il primo errore (impostazione predefinita: 1 s) -**Ritardo massimo**: limite massimo del tempo di attesa (impostazione predefinita: 30 secondi) -**Moltiplicatore**: quanto aumentare il ritardo per guasto consecutivo (impostazione predefinita: 2x)### Anti-thundering herd +Provider profiles support these settings: -Quando molte richieste simultanee raggiungono un provider con velocità limitata, OmniRoute utilizza mutex + limitazione automatica della velocità per serializzare le richieste e prevenire errori a catena. Questo è automatico per i fornitori di chiavi API.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Alcuni utenti OmniRoute posizionano il gateway davanti a RAG o stack di agenti. In queste configurazioni è comune vedere uno schema strano: OmniRoute sembra integro (provider attivi, profili di instradamento ok, nessun avviso di limite di velocità) ma la risposta finale è ancora sbagliata. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -In pratica questi incidenti solitamente provengono dalla pipeline RAG a valle, non dal gateway stesso. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Se desideri un vocabolario condiviso per descrivere questi guasti, puoi utilizzare WFGY ProblemMap, una risorsa di testo con licenza MIT esterna che definisce sedici modelli di fallimento RAG/LLM ricorrenti. Ad alto livello copre: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- deriva del recupero e confini del contesto infranti -- Indici e archivi vettoriali vuoti o obsoleti -- incorporamento e disadattamento semantico -- Problemi relativi all'assemblaggio rapido e alla finestra di contesto -- Collasso logico e risposte troppo sicure -- Errori di coordinamento della catena lunga e degli agenti -- Memoria multiagente e deriva dei ruoli -- Problemi di distribuzione e ordinamento del bootstrap +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -L'idea è semplice: +The idea is simple: -1. Quando investighi su una risposta errata, acquisisci: - - attività e richiesta dell'utente - - Combinazione di percorso o provider in OmniRoute - - qualsiasi contesto RAG utilizzato a valle (documenti recuperati, chiamate strumento, ecc.) -2. Mappare l'incidente su uno o due numeri WFGY ProblemMap (`No.1` … `No.16`). -3. Memorizza il numero nel tuo dashboard, runbook o tracker degli incidenti accanto ai registri OmniRoute. -4. Utilizza la pagina WFGY corrispondente per decidere se è necessario modificare lo stack RAG, il retriever o la strategia di routing. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -Il testo completo e le ricette concrete si trovano qui (licenza MIT, solo testo): +Full text and concrete recipes live here (MIT license, text only): -[README di WFGY ProblemMap](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) +[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -È possibile ignorare questa sezione se non si eseguono RAG o pipeline di agenti dietro OmniRoute.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**Problemi di GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Architettura**: vedere [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) per i dettagli interni -**Riferimento API**: vedere [`docs/API_REFERENCE.md`](API_REFERENCE.md) per tutti gli endpoint -**Dashboard salute**: controlla**Dashboard → Salute**per lo stato del sistema in tempo reale -**Traduttore**: utilizza**Dashboard → Traduttore**per eseguire il debug dei problemi di formato +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/it/llm.txt b/docs/i18n/it/llm.txt new file mode 100644 index 0000000000..b0741f810a --- /dev/null +++ b/docs/i18n/it/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Italiano) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Panoramica + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Sicurezza +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/ja/README.md b/docs/i18n/ja/README.md index e1397da3a0..0aeda24593 100644 --- a/docs/i18n/ja/README.md +++ b/docs/i18n/ja/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_ユニバーサル API プロキシ - 1 つのエンドポイント、60 以上のプロバイダー、ダウンタイムなし。**MCP サーバー (25 ツール)**、**A2A プロトコル**、**メモリ/スキル システム**、**Electron デスクトップ アプリ**が追加されました。_ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**チャット補完 • 埋め込み • 画像生成 • ビデオ • 音楽 • オーディオ • 再ランキング •**Web 検索**• MCP サーバー • A2A プロトコル • 100% TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _ユニバーサル API プロキシ - 1 つのエンドポイント、60 以上 [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 ウェブサイト](https://omniroute.online) • [🚀 クイックスタート](#-quick-start) • [💡 機能](#-key-features) • [📖 ドキュメント](#-documentation) • [💰 価格](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**入手可能な場所:**🇺🇸 [英語](README.md) | 🇧🇷 [ポルトガル語 (ブラジル)](docs/i18n/pt-BR/README.md) | 🇪🇸 [スペイン語](docs/i18n/es/README.md) | 🇫🇷 [フランス語](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [ドイツ語](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [ダンスク](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [マジャル語](docs/i18n/hu/README.md) | 🇮🇩 [インドネシア語](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [メラユ語](docs/i18n/ms/README.md) | 🇳🇱 [オランダ](docs/i18n/nl/README.md) | 🇳🇴 [ノルスク](docs/i18n/no/README.md) | 🇵🇹 [Português (ポルトガル)](docs/i18n/pt/README.md) | 🇷🇴 [ローマ](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [スロベンチナ](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [フィリピン語](docs/i18n/phi/README.md) | 🇨🇿 [チェシュティナ](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,554 +60,629 @@ _ユニバーサル API プロキシ - 1 つのエンドポイント、60 以上 ## 📸 Dashboard Preview -<詳細> +
+Click to see dashboard screenshots -クリックするとダッシュボードのスクリーンショットが表示されます +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | -| ページ | スクリーンショット | -| ------------------ | --------------------------------------------------- | ---------- | -| **プロバイダー** | ![プロバイダー](docs/screenshots/01-providers.png) | -| **コンボ** | ![コンボ](docs/screenshots/02-combos.png) | -| **分析** | ![分析](docs/screenshots/03-analytics.png) | -| **健康** | ![健康](docs/screenshots/04-health.png) | -| **翻訳者** | ![翻訳者](docs/screenshots/05-translator.png) | -| **Settings** | ![設定](docs/screenshots/06-settings.png) | -| **CLI ツール** | ![CLI ツール](docs/screenshots/07-cli-tools.png) | -| **使用ログ** | ![使用法](docs/screenshots/08-usage.png) | -| **エンドポイント** | ![エンドポイント](docs/screenshots/09-endpoint.png) |
| + --- ### 🤖 Free AI Provider for your favorite coding agents -_AI を活用した IDE または CLI ツールを、無制限のコーディングのための無料 API ゲートウェイである OmniRoute 経由で接続します。_ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ -<テーブル> - - - -OpenClaw
-オープンクロー -

-⭐ 205K - - - -NanoBot
-ナノボット -

-⭐ 20.9K - - - -PicoClaw
-ピコクロー -

-⭐ 14.6K - - - -ZeroClaw
-ゼロクロー -

-⭐ 9.9K - - - -IronClaw
-アイアンクロー -

-⭐ 2.1K - - - - - -OpenCode
-オープンコード -

-⭐ 106K - - - -Codex CLI
-コーデックス CLI -

-⭐ 60.8K - - - -クロード コード
-クロード・コード -

-⭐ 67.3K - - - -Gemini CLI
-Gemini CLI -

-⭐ 94.7K - - - -キロコード
-キロコード -

-⭐ 15.5K - - - + + + + + + + + + + + + + + + +
+ + OpenClaw
+ OpenClaw +

+ ⭐ 205K +
+ + NanoBot
+ NanoBot +

+ ⭐ 20.9K +
+ + PicoClaw
+ PicoClaw +

+ ⭐ 14.6K +
+ + ZeroClaw
+ ZeroClaw +

+ ⭐ 9.9K +
+ + IronClaw
+ IronClaw +

+ ⭐ 2.1K +
+ + OpenCode
+ OpenCode +

+ ⭐ 106K +
+ + Codex CLI
+ Codex CLI +

+ ⭐ 60.8K +
+ + Claude Code
+ Claude Code +

+ ⭐ 67.3K +
+ + Gemini CLI
+ Gemini CLI +

+ ⭐ 94.7K +
+ + Kilo Code
+ Kilo Code +

+ ⭐ 15.5K +
-📡 すべてのエージェントは http://localhost:20128/v1 または http://cloud.omniroute.online/v1 経由で接続します - 1 つの構成、無制限のモデルとクォータ--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**お金の無駄遣いや限界に達するのはやめましょう:** +**Stop wasting money and hitting limits:** -- サブスクリプション割り当ては毎月未使用のまま期限切れになります -- レート制限によりコーディングの途中で停止する -- 高価な API (プロバイダーあたり月額 20 ~ 50 ドル) -- プロバイダー間の手動切り替え +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute はこれを解決します:** +**OmniRoute solves this:** -- ✅**サブスクリプションを最大化**- クォータを追跡し、リセットする前にすべてのビットを使用します -- ✅**自動フォールバック**- サブスクリプション → API キー → 安価 → 無料、ダウンタイムなし -- ✅**マルチアカウント**- プロバイダーごとのアカウント間のラウンドロビン -- ✅**ユニバーサル**- Claude Code、Codex、Gemini CLI、Cursor、Cline、OpenClaw、あらゆる CLI ツールで動作します--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**コミュニティに参加してください!**[WhatsApp グループ](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — ヘルプを取得し、ヒントを共有し、最新情報を入手してください。 +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**ウェブサイト**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**問題**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [コミュニティ グループ](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**寄稿**: [CONTRIBUTING.md](CONTRIBUTING.md) を参照するか、PR を開くか、「良い最初の号」を選択してください。-**オリジナルプロジェクト**: [9router by decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -問題を開くときは、system-info コマンドを実行して、生成されたファイルを添付してください。```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -これにより、Node.js バージョン、OmniRoute バージョン、OS の詳細、インストールされている CLI ツール (qoder、gemini、claude、codex、antigravity、droid など)、Docker/PM2 ステータス、およびシステム パッケージを含む `system-info.txt` が生成されます。これには、問題を迅速に再現するために必要なものがすべて含まれています。ファイルを GitHub の問題に直接添付します。--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**AI ツールを使用するすべての開発者は、これらの問題に日々直面しています。**OmniRoute は、コスト超過から地域ブロック、壊れた OAuth フローからプロトコル操作、企業の可観測性まで、それらすべてを解決するために構築されました。 +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. -<詳細> -💸 1. 「高額なサブスクリプションの料金を支払っているのに、制限によって中断される」 +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -開発者は、Claude Pro、Codex Pro、または GitHub Copilot に月額 20 ~ 200 ドルを支払います。有料であっても、割り当てには上限があり、5 時間の使用量、週ごとの制限、または分ごとのレート制限があります。コーディング セッションの途中でプロバイダーが応答を停止し、開発者はフローと生産性を失います。 +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**OmniRoute がそれを解決する方法:** +**How OmniRoute solves it:** --**スマート 4 層フォールバック**— サブスクリプション クォータが不足すると、手動介入なしで API キー→格安→無料に自動的にリダイレクトされます。 --**プロバイダー制限の追跡**- キャッシュされたクォータ スナップショットはサーバー側のスケジュール (デフォルトは「PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70」) で更新され、UI で手動更新が可能になります。 --**マルチアカウントのサポート**— 自動ラウンドロビンによるプロバイダーごとの複数のアカウント — 1 つのアカウントがなくなると、次のアカウントに切り替わります --**カスタム コンボ**— 9 つのバランス戦略 (優先順位、重み付け、フィルファースト、ラウンドロビン、P2C、ランダム、最も使用されていない、コスト最適化、厳密ランダム) を備えたカスタマイズ可能なフォールバック チェーン --**Codex Business クォータ**— ビジネス/チームのワークスペース クォータをダッシュボードで直接監視
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard -<詳細> -🔌 2. 「複数のプロバイダーを使用する必要があるが、それぞれに異なる API がある」 + -OpenAI は 1 つの形式を使用し、Claude (Anthropic) は別の形式を使用し、Gemini はさらに別の形式を使用します。開発者が異なるプロバイダーのモデルをテストしたり、プロバイダー間でフォールバックしたりする場合は、SDK を再構成し、エンドポイントを変更し、互換性のない形式に対処する必要があります。カスタム プロバイダー (FriendLI、NIM) には、非標準モデルのエンドポイントがあります。 +
+🔌 2. "I need to use multiple providers but each has a different API" -**OmniRoute がそれを解決する方法:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**統合エンドポイント**— 単一の「http://localhost:20128/v1」が 60 を超えるすべてのプロバイダーのプロキシとして機能します --**フォーマット変換**— 自動かつ透過的: OpenAI ↔ Claude ↔ Gemini ↔ Responses API --**レスポンスのサニタイズ**— OpenAI SDK v1.83+ を破壊する非標準フィールド (`x_groq`、`usage_breakdown`、`service_tier`) を削除します。 --**ロールの正規化**— 非 OpenAI プロバイダーの場合は「開発者」→「システム」に変換します。 GLM/ERNIEの場合は「システム」→「ユーザー」 --**Think Tag Extraction**— DeepSeek R1 などのモデルから `` ブロックを標準化された `reasoning_content` に抽出します。 --**Gemini の構造化出力**— `json_schema` → `responseMimeType`/`responseSchema` の自動変換 --**`stream` のデフォルトは `false`**— OpenAI 仕様に準拠し、Python/Rust/Go SDK での予期しない SSE を回避します
+**How OmniRoute solves it:** -<詳細> -🌐 3. 「AI プロバイダーが私の地域/国をブロックしています」 +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -OpenAI/Codex などのプロバイダーは、特定の地理的地域からのアクセスをブロックします。ユーザーは、OAuth および API 接続中に「unsupported_country_region_territory」のようなエラーを受け取ります。これは、発展途上国の開発者にとって特にイライラさせられます。 + -**OmniRoute がそれを解決する方法:** +
+🌐 3. "My AI provider blocks my region/country" --**3 レベルのプロキシ構成**— 3 つのレベルで構成可能なプロキシ: グローバル (すべてのトラフィック)、プロバイダーごと (1 つのプロバイダーのみ)、および接続/キーごと --**色分けされたプロキシ バッジ**— 視覚的なインジケーター: 🟢 グローバル プロキシ、🟡 プロバイダー プロキシ、🔵 接続プロキシ、常に IP を表示 --**プロキシを介した OAuth トークン交換**— OAuth フローもプロキシを通過し、「unsupported_country_region_territory」を解決します --**プロキシ経由の接続テスト**— 接続テストは構成されたプロキシを使用します (直接バイパスはありません)。 --**SOCKS5 サポート**— アウトバウンド ルーティングに対する SOCKS5 プロキシの完全なサポート --**TLS フィンガープリント スプーフィング**— 「wreq-js」を介したブラウザーのような TLS フィンガープリントによりボット検出をバイパスします --**🔏 CLI フィンガープリント マッチング**— ネイティブ CLI バイナリ署名と一致するようにヘッダーと本文フィールドの順序を変更し、アカウントにフラグを立てるリスクを大幅に軽減します。プロキシ IP は保持されます。ステルス**と**IP マスキングの両方を同時に取得できます。
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. -<詳細> -🆓 4. 「コーディングに AI を使用したいが、お金がない」 +**How OmniRoute solves it:** -誰もが AI サブスクリプションに月額 20 ~ 200 ドルを支払えるわけではありません。学生、新興国の開発者、愛好家、フリーランサーは、高品質のモデルに無料でアクセスできる必要があります。 +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**OmniRoute がそれを解決する方法:** + --**無料利用枠プロバイダーの組み込み**— 100% 無料プロバイダーのネイティブ サポート: Qoder (OAuth 経由の 5 つの無制限のモデル: kimi-k2- Thinking、qwen3-coder-plus、deepseek-r1、minimax-m2、kimi-k2)、Qwen (4 つの無制限のモデル: qwen3-coder-plus、qwen3-coder-flash、qwen3-coder-next、 vision-model)、Kiro (Claude + AWS Builder ID は無料)、Gemini CLI (180,000 トークン/月は無料) --**Ollama Cloud**— `api.ollama.com` にあるクラウド ホスト型の Ollama モデル (無料の「ライト使用量」レベル)。 `ollamacloud/` プレフィックスを使用します --**無料のみのコンボ**— チェーン `gc/gemini-3-flash → if/kimi-k2- Thinking → qw/qwen3-coder-plus` = 月額 0 ドル、ダウンタイムなし --**NVIDIA NIM 無料アクセス**— ~40 RPM 開発 - build.nvidia.com で 70 以上のモデルに永久に無料アクセス (クレジットから純粋なレート制限に移行) --**コスト最適化戦略**— 利用可能な最も安価なプロバイダーを自動的に選択するルーティング戦略 +
+🆓 4. "I want to use AI for coding but I have no money" -<詳細> -🔒 5. 「AI ゲートウェイを不正アクセスから保護する必要がある」 +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -AI ゲートウェイをネットワーク (LAN、VPS、Docker) に公開すると、アドレスを持っている人は誰でも開発者のトークン/クォータを消費できます。保護がなければ、API は誤用、即時挿入、悪用に対して脆弱になります。 +**How OmniRoute solves it:** -**OmniRoute がそれを解決する方法:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**API キー管理**— 専用の `/dashboard/api-manager` ページを使用したプロバイダーごとの生成、ローテーション、スコープ設定 --**モデルレベルの権限**— すべて許可/制限の切り替えにより、API キーを特定のモデル (`openai/*`、ワイルドカード パターン) に制限します --**API エンドポイント保護**— `/v1/models` のキーを要求し、リストから特定のプロバイダーをブロックします --**認証ガード + CSRF 保護**— すべてのダッシュボード ルートは「withAuth」ミドルウェア + CSRF トークンで保護されています --**レート リミッター**— 構成可能なウィンドウによる IP ごとのレート制限 --**IP フィルタリング**— アクセス制御の許可リスト/ブロックリスト --**プロンプト インジェクション ガード**— 悪意のあるプロンプト パターンに対するサニタイズ --**AES-256-GCM 暗号化**— 認証情報は保存時に暗号化されます
+ -<詳細> -🛑 6. 「プロバイダーがダウンし、コーディング フローが失われてしまいました。」 +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -AI プロバイダーが不安定になったり、5xx エラーを返したり、一時的なレート制限に達したりする可能性があります。開発者が単一のプロバイダーに依存している場合、それらは中断されます。サーキット ブレーカーがないと、再試行を繰り返すとアプリケーションがクラッシュする可能性があります。 +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**OmniRoute がそれを解決する方法:** +**How OmniRoute solves it:** --**モデルごとのサーキット ブレーカー**— 設定可能なしきい値とクールダウン (閉/開/半開) による自動開閉、カスケード ブロックを回避するためのモデルごとのスコープ --**指数バックオフ**— 漸進的な再試行遅延 --**Anti-Thundering Herd**— 同時再試行の嵐に対するミューテックス + セマフォ保護 --**コンボ フォールバック チェーン**- プライマリ プロバイダーに障害が発生した場合、介入なしで自動的にチェーンを通過します。 --**コンボ サーキット ブレーカー**— コンボ チェーン内の障害が発生したプロバイダーを自動的に無効にします --**ヘルス ダッシュボード**— 稼働時間モニタリング、サーキット ブレーカーの状態、ロックアウト、キャッシュ統計、p50/p95/p99 レイテンシ
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest -<詳細> -🔧 7. 「各 AI ツールの設定は面倒で繰り返しが多い」 + -開発者は、Cursor、Claude Code、Codex CLI、OpenClaw、Gemini CLI、Kilo Code を使用します。各ツールには異なる構成 (API エンドポイント、キー、モデル) が必要です。プロバイダーや機種変更時に再設定するのは時間の無駄です。 +
+🛑 6. "My provider went down and I lost my coding flow" -**OmniRoute がそれを解決する方法:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**CLI ツール ダッシュボード**— Claude Code、Codex CLI、OpenClaw、Kilo Code、Antigravity、Cline をワンクリックでセットアップできる専用ページ --**GitHub Copilot Config Generator**— モデルを一括選択して VS Code の `chatLanguageModels.json` を生成します --**オンボーディング ウィザード**- 初めてユーザー向けのガイド付き 4 ステップ セットアップ --**1 つのエンドポイント、すべてのモデル**- `http://localhost:20128/v1` を一度構成すると、60 を超えるプロバイダーにアクセスできます
+**How OmniRoute solves it:** -<詳細> -🔑 8. 「複数のプロバイダーからの OAuth トークンを管理するのは地獄です」 +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code、Codex、Gemini CLI、Copilot — すべては有効期限切れのトークンを持つ OAuth 2.0 を使用します。開発者は定期的に再認証し、「client_secret is missing」、「redirect_uri_mismatch」、およびリモートサーバー上の障害に対処する必要があります。 LAN/VPS 上の OAuth は特に問題があります。 + -**OmniRoute がそれを解決する方法:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**自動トークン更新**— OAuth トークンは有効期限が切れる前にバックグラウンドで更新されます。 --**OAuth 2.0 (PKCE) ビルトイン**— Claude Code、Codex、Gemini CLI、Copilot、Kiro、Qwen、Qoder の自動フロー --**マルチアカウント OAuth**— JWT/ID トークン抽出によるプロバイダーごとの複数のアカウント --**OAuth LAN/リモート修正**— `redirect_uri` のプライベート IP 検出 + リモート サーバーの手動 URL モード --**Nginx の背後にある OAuth**— リバース プロキシの互換性のために `window.location.origin` を使用します --**リモート OAuth ガイド**— VPS/Docker での Google Cloud 認証情報のステップバイステップ ガイド
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. -<詳細> -📊 9. 「どこにいくら使っているのか分かりません」 +**How OmniRoute solves it:** -開発者は複数の有料プロバイダーを使用していますが、支出について統一した見解がありません。各プロバイダーには独自の請求ダッシュボードがありますが、統合されたビューはありません。予期せぬ出費がかさむ可能性があります。 +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**OmniRoute がそれを解決する方法:** + --**コスト分析ダッシュボード**— トークンごとのコスト追跡とプロバイダーごとの予算管理 --**階層ごとの予算制限**— 自動フォールバックをトリガーする階層ごとの支出上限 --**モデルごとの価格構成**— モデルごとに構成可能な価格 --**API キーごとの使用統計**— キーごとのリクエスト数と最後に使用されたタイムスタンプ --**分析ダッシュボード**— 統計カード、モデル使用状況グラフ、成功率と遅延を含むプロバイダー表 +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" -<詳細> -🐛 10. 「AI 呼び出しのエラーや問題を診断できない」 +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -呼び出しが失敗すると、開発者はそれがレート制限なのか、トークンの期限切れなのか、間違った形式なのか、プロバイダーのエラーなのかわかりません。異なる端末間で断片化されたログ。可観測性がなければ、デバッグは試行錯誤になります。 +**How OmniRoute solves it:** -**OmniRoute がそれを解決する方法:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**統合ログ ダッシュボード**— 4 つのタブ: リクエスト ログ、プロキシ ログ、監査ログ、コンソール --**コンソール ログ ビューア**— 色分けされたレベル、自動スクロール、検索、フィルターを備えたリアルタイムのターミナル スタイルのビューア --**SQLite プロキシ ログ**— サーバーの再起動後も存続する永続的なログ --**Translator Playground**— 4 つのデバッグ モード: プレイグラウンド (形式変換)、チャット テスター (往復)、テストベンチ (バッチ)、ライブ モニター (リアルタイム) --**リクエスト テレメトリ**— p50/p95/p99 レイテンシ + X-Request-Id トレース --**ファイルベースのローテーションによるログ**- アプリのログは、サイズ、保存日数、アーカイブ数に基づいてローテーションされます。通話ログのアーティファクトは保存日数とファイル数によってローテーションされます --**システム情報レポート**- 「npm run system-info」は、完全な環境 (ノードのバージョン、OmniRoute のバージョン、OS、CLI ツール、Docker/PM2 ステータス) を含む「system-info.txt」を生成します。即時トリアージのために問題を報告するときに添付してください。
+ -<詳細> -🏗️ 11. 「ゲートウェイの導入と保守は複雑です」 +
+📊 9. "I don't know how much I'm spending or where" -さまざまな環境 (ローカル、VPS、Docker、クラウド) 間で AI プロキシをインストール、構成、維持するには、多大な労力がかかります。ハードコードされたパス、ディレクトリの「EACCES」、ポートの競合、クロスプラットフォーム ビルドなどの問題により、摩擦が増大します。 +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**OmniRoute がそれを解決する方法:** +**How OmniRoute solves it:** --**npm global install**— `npm install -gomniroute &&omniroute` — 完了 --**Docker マルチプラットフォーム**— AMD64 + ARM64 ネイティブ (Apple Silicon、AWS Graviton、Raspberry Pi) --**Docker Compose プロファイル**— `base` (CLI ツールなし) および `cli` (Claude Code、Codex、OpenClaw を含む) --**Electron デスクトップ アプリ**— システム トレイ、自動起動、オフライン モードを備えた Windows/macOS/Linux 用ネイティブ アプリ --**分割ポート モード**- 高度なシナリオ (リバース プロキシ、コンテナ ネットワーキング) 向けに個別のポート上の API とダッシュボード --**Cloud Sync**— Cloudflare Workers を介したデバイス間での設定の同期 --**DB バックアップ**— 外部管理バックアップ用の `DISABLE_SQLITE_AUTO_BACKUP` を使用した、すべての設定の自動バックアップ、復元、エクスポートおよびインポート
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency -<詳細> -🌍 12. 「インターフェースは英語のみで、私のチームは英語を話せません。」 + -非英語圏の国、特にラテンアメリカ、アジア、ヨーロッパのチームは、英語のみのインターフェースに苦労しています。言語の壁があると採用が減り、構成エラーが増加します。 +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**OmniRoute がそれを解決する方法:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**ダッシュボード i18n — 30 言語**— アラビア語、ブルガリア語、デンマーク語、ドイツ語、スペイン語、フィンランド語、フランス語、ヘブライ語、ヒンディー語、ハンガリー語、インドネシア語、イタリア語、日本語、韓国語、マレー語、オランダ語、ノルウェー語、ポーランド語、ポルトガル語 (PT/BR)、ルーマニア語、ロシア語、スロバキア語、スウェーデン語、タイ語、ウクライナ語、ベトナム語、中国語、フィリピン語、英語 --**RTL サポート**— アラビア語とヘブライ語の右から左へのサポート --**Multi-Language READMEs**— 30 complete documentation translations --**言語セレクター**— リアルタイム切り替えのためのヘッダーの地球儀アイコン
+**How OmniRoute solves it:** -<詳細> -🔄 13. 「チャット以上のものが必要です - 埋め込み、画像、音声が必要です」 +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -AIは単なるチャット補完ではありません。開発者は、画像の生成、音声の文字起こし、RAG の埋め込みの作成、ドキュメントの再ランク付け、およびコンテンツの管理を行う必要があります。各 API には異なるエンドポイントと形式があります。 + -**OmniRoute がそれを解決する方法:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Embeddings**— 6 つのプロバイダーと 9 つ以上のモデルを含む `/v1/embeddings` --**画像生成**— 10 のプロバイダーと 20 以上のモデル (OpenAI、xAI、Togetter、Fireworks、Nebius、Hyperbolic、NanoBanana、Antigravity、SD WebUI、ComfyUI) を含む `/v1/images/generations` --**Text-to-Video**— `/v1/videos/generations` — ComfyUI (AnimateDiff、SVD) および SD WebUI --**Text-to-Music**— `/v1/music/generations` — ComfyUI (Stable Audio Open、MusicGen) --**音声転写**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM、HuggingFace、Qwen3 --**テキスト読み上げ**— `/v1/audio/speech` — イレブンラボ、Nvidia NIM、HuggingFace、Coqui、Tortoise、Qwen3、**Inworld**、**Cartesia**、**PlayHT**、+ 既存のプロバイダー --**モデレーション**— `/v1/moderations` — コンテンツの安全性チェック --**再ランキング**— `/v1/rerank` — ドキュメントの関連性の再ランキング --**Responses API**— Codex の完全な `/v1/responses` サポート
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. -<詳細> -🧪 14. 「モデル間で品質をテストして比較する方法がない」 +**How OmniRoute solves it:** -開発者は、コード、翻訳、推論などのユースケースにどのモデルが最適であるかを知りたいと考えていますが、手動で比較するのは時間がかかります。統合された評価ツールは存在しません。 +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**OmniRoute がそれを解決する方法:** + --**LLM 評価**— 挨拶、数学、地理、コード生成、JSON 準拠、翻訳、マークダウン、安全性の拒否をカバーする 10 個の事前ロードされたケースによるゴールデン セット テスト --**4 つの一致戦略**— `exact`、`contains`、`regex`、`custom` (JS 関数) --**Translator Playground Test Bench**— 複数の入力と予想される出力を使用したバッチ テスト、クロスプロバイダー比較 --**チャット テスター**— 視覚的な応答レンダリングによる完全な往復 --**ライブ モニター**— プロキシを通過するすべてのリクエストのリアルタイム ストリーム +
+🌍 12. "The interface is English-only and my team doesn't speak English" -<詳細> -📈 15. 「パフォーマンスを落とさずにスケールする必要がある」 +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -リクエストの量が増えると、同じ質問をキャッシュしないと重複したコストが発生します。冪等性がないと、重複したリクエストにより処理が無駄になります。プロバイダーごとのレート制限を遵守する必要があります。 +**How OmniRoute solves it:** -**OmniRoute がそれを解決する方法:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**セマンティック キャッシュ**— 2 層キャッシュ (署名 + セマンティック) によりコストと遅延が削減されます。 --**リクエストの冪等性**— 同一のリクエストに対する重複排除ウィンドウは 5 秒です --**レート制限検出**— プロバイダーごとの RPM、最小ギャップ、最大同時トラッキング --**編集可能なレート制限**— [設定] → [永続性を伴う回復力] で構成可能なデフォルト --**API キー検証キャッシュ**— 運用パフォーマンスのための 3 層キャッシュ --**テレメトリ付きヘルス ダッシュボード**— p50/p95/p99 レイテンシ、キャッシュ統計、稼働時間
+ -<詳細> -🤖 16. 「モデルの動作をグローバルに制御したい」 +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -すべての応答を特定の言語、特定の口調で行いたい、または推論トークンを制限したい開発者。すべてのツール/リクエストでこれを設定するのは現実的ではありません。 +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**OmniRoute がそれを解決する方法:** +**How OmniRoute solves it:** --**システム プロンプト インジェクション**— すべてのリクエストに適用されるグローバル プロンプト --**思考予算検証**— リクエストごとの推論トークン割り当て制御 (パススルー、自動、カスタム、適応) --**9 ルーティング戦略**— リクエストの分散方法を決定するグローバル戦略 --**ワイルドカード ルーター**— `provider/*` パターンは任意のプロバイダーに動的にルーティングします。 --**コンボ有効/無効切り替え**— ダッシュボードから直接コンボを切り替えます --**プロバイダー切り替え**— ワンクリックでプロバイダーのすべての接続を有効/無効にします。 --**ブロックされたプロバイダー**— `/v1/models` リストから特定のプロバイダーを除外します
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex -<詳細> -🧰 17. 「一流の製品機能として MCP ツールが必要です」 + -多くの AI ゲートウェイは、MCP を非表示の実装詳細としてのみ公開します。チームには、目に見えて管理しやすいオペレーション レイヤーが必要です。 +
+🧪 14. "I have no way to test and compare quality across models" -**OmniRoute がそれを解決する方法:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP はダッシュボードのナビゲーションとエンドポイント プロトコル タブに表示されます -- プロセス、ツール、スコープ、監査を備えた専用の MCP 管理ページ -- 「omniroute --mcp」およびクライアントのオンボーディング用の組み込みクイックスタート
+**How OmniRoute solves it:** -<詳細> -🧠 18. 「同期 + ストリーム タスク パスを備えた A2A オーケストレーションが必要です」 +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -エージェント ワークフローには、直接応答と、ライフサイクル制御による長時間実行のストリーミング実行の両方が必要です。 + -**OmniRoute がそれを解決する方法:** +
+📈 15. "I need to scale without losing performance" -- A2A JSON-RPC エンドポイント (`POST /a2a`) と `message/send` および `message/stream` -- 端末状態の伝播を伴う SSE ストリーミング -- `tasks/get` および `tasks/cancel` のタスク ライフサイクル API
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. -<詳細> -🛰️ 19. 「推測されたステータスではなく、実際の MCP プロセスの健全性が必要です」 +**How OmniRoute solves it:** -運用チームは、API が到達可能かどうかだけでなく、MCP が実際に生きているかどうかを知る必要があります。 +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**OmniRoute がそれを解決する方法:** + -- PID、タイムスタンプ、トランスポート、ツール数、およびスコープモードを含むランタイムハートビートファイル -- ハートビートと最近のアクティビティを組み合わせた MCP ステータス API -- プロセス/稼働時間/ハートビートの鮮度を示す UI ステータス カード +
+🤖 16. "I want to control model behavior globally" -<詳細> -📋 20. 「監査可能な MCP ツールの実行が必要です」 +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -ツールが構成を変更したり、運用アクションをトリガーしたりする場合、チームはフォレンジックなトレーサビリティを必要とします。 +**How OmniRoute solves it:** -**OmniRoute がそれを解決する方法:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- MCP ツール呼び出しの SQLite ベースの監査ログ -- ツール、成功/失敗、API キー、ページネーションによるフィルター -- ダッシュボード監査テーブル + 自動化のための統計エンドポイント
+ -<詳細> -🔐 21. 「統合ごとにスコープ指定された MCP 権限が必要です」 +
+🧰 17. "I need MCP tools as first-class product capabilities" -異なるクライアントには、ツール カテゴリへの最小限の特権アクセスが必要です。 +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**OmniRoute がそれを解決する方法:** +**How OmniRoute solves it:** -- 制御されたツールアクセスのための 10 個の詳細な MCP スコープ -- MCP 管理 UI でのスコープの適用と可視性 -- 運用ツールの安全なデフォルト姿勢
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding -<詳細> -⚙️ 22. 「再展開せずに運用管理が必要です」 + -チームは、インシデントやコスト イベントが発生した際に、実行時の変更を迅速に行う必要があります。 +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**OmniRoute がそれを解決する方法:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- MCP ダッシュボードからコンボのアクティブ化を直接切り替えます -- 事前定義されたポリシーパックから復元プロファイルを適用 -- 同じ操作パネルからサーキットブレーカーの状態をリセット
+**How OmniRoute solves it:** -<詳細> -🔄 23. 「ライブ A2A タスクのライフサイクルの可視化とキャンセルが必要です」 +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -ライフサイクルの可視性がなければ、タスク インシデントの優先順位付けが困難になります。 + -**OmniRoute がそれを解決する方法:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- ページネーションを使用した状態/スキルによるタスクのリスト/フィルタリング -- タスクのメタデータ、イベント、アーティファクトのドリルダウン -- タスクキャンセルエンドポイントと確認付きの UI アクション
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. -<詳細> -🌊 24. 「A2A ロード用のアクティブ ストリーム メトリクスが必要です」 +**How OmniRoute solves it:** -ストリーミング ワークフローには、同時実行性とライブ接続に関する運用上の洞察が必要です。 +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**OmniRoute がそれを解決する方法:** + -- アクティブ ストリーム カウンターが A2A ステータスに統合されました -- 最後のタスクのタイムスタンプと状態ごとのカウント -- リアルタイム運用監視用の A2A ダッシュボード カード +
+📋 20. "I need auditable MCP tool execution" -<詳細> -🪪 25. 「クライアント用の標準エージェント検出が必要です」 +When tools mutate config or trigger ops actions, teams need forensic traceability. -外部クライアントとオーケストレーターには、オンボーディング用の機械可読メタデータが必要です。 +**How OmniRoute solves it:** -**OmniRoute がそれを解決する方法:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- エージェント カードは `/.well-known/agent.json` で公開されます -- 管理UIに表示される能力とスキル -- A2A ステータス API には自動化のための検出メタデータが含まれています
+ -<詳細> -🧭 26. 「製品の UX にプロトコルの検出機能が必要です」 +
+🔐 21. "I need scoped MCP permissions per integration" -ユーザーがプロトコルの表面を発見できない場合、導入とサポートの品質が低下します。 +Different clients should have least-privilege access to tool categories. -**OmniRoute がそれを解決する方法:** +**How OmniRoute solves it:** -- プロキシ、MCP、A2A、API エンドポイントのタブを備えた統合された**エンドポイント**ページ -- MCP および A2A のインライン サービス ステータスの切り替え (オンライン/オフライン) -- 概要から専用の管理タブへのリンク
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling -<詳細> -🧪 27. 「実際のクライアントを使用したエンドツーエンドのプロトコル検証が必要です」 + -模擬テストは、リリース前にプロトコルの互換性を検証するには十分ではありません。 +
+⚙️ 22. "I need operational controls without redeploying" -**OmniRoute がそれを解決する方法:** +Teams need quick runtime changes during incidents or cost events. -- アプリを起動し、実際の MCP SDK クライアント トランスポートを使用する E2E スイート -- A2A クライアントは、フローの検出、送信、ストリーミング、取得、キャンセルをテストします。 -- MCP 監査および A2A タスク API に対するアサーションのクロスチェック
+**How OmniRoute solves it:** -<詳細> -📡 28. 「すべてのインターフェースにわたって統合された可観測性が必要です」 +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -プロトコルごとに可観測性を分割すると、盲点が生じ、MTTR が長くなります。 + -**OmniRoute がそれを解決する方法:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- ダッシュボード/ログ/分析を 1 つの製品に統合 -- OpenAI、MCP、A2A レイヤーにわたるヘルス + 監査 + リクエスト テレメトリ -- ステータスと自動化のための運用 API
+Without lifecycle visibility, task incidents become hard to triage. -<詳細> -💼 29. 「プロキシ + ツール + エージェント オーケストレーション用に 1 つのランタイムが必要です」 +**How OmniRoute solves it:** -多くの個別のサービスを実行すると、運用コストが増加し、障害モードが増加します。 +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**OmniRoute がそれを解決する方法:** + -- OpenAI互換プロキシ、MCPサーバー、A2Aサーバーを1つのスタックに搭載 -- 共有認証、復元力、データストア、可観測性 -- すべての対話面にわたる一貫したポリシー モデル +
+🌊 24. "I need active stream metrics for A2A load" -<詳細> -🚀 30. 「グルーコードのスプロールを発生させずにエージェント ワークフローを出荷する必要がある」 +Streaming workflows require operational insight into concurrency and live connections. -複数のアドホック サービスとスクリプトをつなぎ合わせると、チームの速度が低下します。 +**How OmniRoute solves it:** -**OmniRoute がそれを解決する方法:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- クライアントとエージェント向けの統合エンドポイント戦略 -- 組み込みのプロトコル管理 UI とスモーク検証パス -- 本番環境に対応した基盤 (セキュリティ、ロギング、復元力、バックアップ)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**戦略 A: 有料サブスクリプション + 安価なバックアップを最大限に活用する**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -608,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**プレイブック B: ゼロコストのコーディング スタック**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**プレイブック C: 24 時間年中無休の常時オンのフォールバック チェーン**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -631,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**プレイブック D: MCP + A2A を使用したエージェントの運用**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost ->**$0/月**で AI コーディングを数分でセットアップできます。これらの無料アカウントを接続し、組み込みの**Free Stack**コンボを使用します。 +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -|ステップ |アクション |プロバイダーのロックが解除されました | +| Step | Action | Providers Unlocked | | ---- | -------------------------------------------------- | ------------------------------------------------------------------ | -| 1 |**Kiro**に接続 (AWS ビルダー ID OAuth) | Claude Sonnet 4.5、Haiku 4.5 —**無制限**| -| 2 |**Qoder**に接続 (Google OAuth) | kimi-k2- Thinking、qwen3-coder-plus、deepseek-r1... —**無制限**| -| 3 |**Qwen**を接続 (デバイス コード) | qwen3-coder-plus、qwen3-coder-flash... —**無制限**| -| 4 |**Gemini CLI**に接続する (Google OAuth) | gemini-3-flash、gemini-2.5-pro —**180K/月無料**| -| 5 | `/dashboard/combos` →**無料スタック ($0)**テンプレート |すべての無料プロバイダーを自動的にラウンドロビンします。 +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**任意の IDE/CLI を次のように指定します。**`http://localhost:20128/v1` · API キー: `any-string` · 完了。 +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**オプションの追加補償範囲 (こちらも無料):**Groq API キー (30 RPM 無料)、NVIDIA NIM (40 RPM 無料、70 以上のモデル)、Cerebras (100 万トークン/日)、LongCat API キー (5000 万トークン/日!)、Cloudflare Workers AI (10,000 ニューロン/日、50 以上のモデル)。## クイックスタート +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## クイックスタート ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **pnpm ユーザー:**インストール後に `pnpmapprove-builds -g` を実行して、`better-sqlite3` および `@swc/core` に必要なネイティブ ビルド スクリプトを有効にします。 +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > -> 「バッシュ」 -> pnpm install -g オムニルート -> pnpm accept-builds -g # すべてのパッケージを選択 → 承認 -> オムニルート -> 「」 +> ```bash +> pnpm install -g omniroute +> pnpm approve-builds -g # Select all packages → approve +> omniroute +> ``` Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| コマンド | 説明 | -| ---------------------------- | --------------------------------------------------------------------------------- | -| `オムニルート` | サーバーを起動します (`PORT=20128`、API とダッシュボードは同じポート上にあります) | -| `オムニルート --ポート 3000` | 正規/API ポートを 3000 に設定します。 | -| `omniroute --mcp` | MCP サーバー (stdio トランスポート) を開始します。 | -| `omniroute --no-open` | ブラウザを自動的に開かない | -| `omniroute --help` | ヘルプを表示 | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -オプションの分割ポート モード:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -ほとんどの展開では、次のものだけが必要です。 +For most deployments, you only need: -|変数 |デフォルト |目的 | +| Variable | Default | Purpose | | ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600000` |アップストリームフェッチ、非表示の Undici タイムアウト、TLS フィンガープリントリクエスト、API ブリッジリクエスト/プロキシタイムアウトの共有ベースライン | -| `STREAM_IDLE_TIMEOUT_MS` | `REQUEST_TIMEOUT_MS` を継承します。 OmniRoute が SSE ストリームを中止するまでのストリーミング チャンク間の最大ギャップ | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -下位互換性は維持されます。既存の `FETCH_TIMEOUT_MS`、`API_BRIDGE_PROXY_TIMEOUT_MS`、およびその他のレイヤーごとのタイムアウト変数は引き続き機能し、共有ベースラインをオーバーライドします。 +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Advanced overrides are available if you need finer control:|変数 |デフォルト |目的 | -| -------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | `REQUEST_TIMEOUT_MS` を継承します。メインのフェッチ中止信号によって使用されるアップストリーム要求タイムアウトの合計 | -| `FETCH_HEADERS_TIMEOUT_MS` | `FETCH_TIMEOUT_MS` を継承します。アップストリーム応答ヘッダーを受信するための Undici 時間制限 | -| `FETCH_BODY_TIMEOUT_MS` | `FETCH_TIMEOUT_MS` を継承します。上流の本体チャンク間の Undici 時間制限 (「0」はそれを無効にします) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP 接続タイムアウト | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici アイドル キープアライブ ソケット タイムアウト | -| `TLS_CLIENT_TIMEOUT_MS` | `FETCH_TIMEOUT_MS` を継承します。 「wreq-js」を通じて行われた TLS フィンガープリント要求のタイムアウト | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | `REQUEST_TIMEOUT_MS` または `30000` を継承します。 API ポートからダッシュボード ポートへの「/v1」プロキシ転送のタイムアウト | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | API ブリッジ サーバーでの受信リクエストのタイムアウト | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | API ブリッジ サーバーでの受信ヘッダーのタイムアウト | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | API ブリッジ サーバーでのキープアライブ タイムアウト | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | API ブリッジ サーバーのソケット非アクティブ タイムアウト (「0」は無効にします) | +Advanced overrides are available if you need finer control: -Nginx、Caddy、Cloudflare、または別のリバース プロキシの背後で OmniRoute を実行している場合は、プロキシが -タイムアウトも、OmniRoute のストリーム/フェッチ タイムアウトよりも長くなります。### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. ダッシュボード → 「プロバイダー」を開き、少なくとも 1 つのプロバイダー (OAuth または API キー) に接続します。 -2. 「ダッシュボード」→「エンドポイント」を開き、API キーを作成します。 -3. (オプション) [ダッシュボード] → [コンボ] を開き、フォールバック チェーンを設定します。### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Claude Code、Codex CLI、Gemini CLI、Cursor、Cline、OpenClaw、OpenCode、OpenAI 互換 SDK で動作します。### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (ツール主導の操作用):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -次に、MCP クライアントを「stdio」経由で接続し、次のようなツールをテストします。 +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (エージェント間のワークフロー用):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -760,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -このスイートは、実行中のアプリに対して実際の MCP および A2A クライアント フローを検証します。### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -768,14 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` -<詳細> +
+Void Linux (`xbps-src` template) -Void Linux (`xbps-src` テンプレート) - -Void Linux ユーザーの場合は、「xbps-src」を使用してネイティブ パッケージを構築できます。このブロックを `srcpkgs/omniroute/template` として保存します。```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -787,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -795,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -871,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -882,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute は、[Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute) でパブリック Docker イメージとして利用できます。 +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**クイック実行:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -892,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**環境ファイルあり:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Docker Compose の使用:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Docker デプロイメントのダッシュボードのサポートには、「ダッシュボード → エンドポイント」のワンクリック**Cloudflare Quick Tunnel**が含まれるようになりました。最初の有効化では、必要な場合にのみ「cloudflared」をダウンロードし、現在の「/v1」エンドポイントへの一時トンネルを開始し、生成された「https://\*.trycloudflare.com/v1」 URL を通常のパブリック URL の直下に表示します。 +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -注: +Notes: -- クイック トンネル URL は一時的なもので、再起動するたびに変更されます。 -- クイック トンネルは、OmniRoute またはコンテナの再起動後に自動復元されません。必要に応じて、ダッシュボードから再度有効にします。 -- マネージド インストールは現在、`x64` / `arm64` 上の Linux、macOS、および Windows をサポートしています。 -- マネージド クイック トンネルは、制約のあるコンテナ環境でのノイズの多い QUIC UDP バッファ警告を回避するために、デフォルトで HTTP/2 トランスポートになります。別のトランスポートが必要な場合は、「CLOUDFLARED_PROTOCOL=quic」または「auto」を設定します。 -- Docker イメージはシステム CA ルートをバンドルし、管理対象の「cloudflared」に渡します。これにより、トンネルがコンテナ内でブートストラップするときの TLS 信頼障害が回避されます。 -- SQLite は WAL モードで実行されます。 OmniRoute が最新の変更をチェックポイントで「storage.sqlite」に戻すことができるように、「docker stop」の終了を許可する必要があります。 -- バンドルされている Compose ファイルには、すでに 40 秒の停止猶予期間が設定されています。イメージを直接実行する場合は、手動停止によってシャットダウンのクリーンアップが中断されないように、「--stop-timeout 40」 (または同様のもの) を維持してください。 -- OmniRoute でバイナリをダウンロードする代わりに既存のバイナリを使用する場合は、「CLOUDFLARED_BIN=/absolute/path/to/cloudflared」を設定します。 +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Caddy での Docker Compose の使用 (HTTPS 自動 TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute は、Caddy の自動 SSL プロビジョニングを使用して安全に公開できます。ドメインの DNS A レコードがサーバーの IP を指していることを確認してください。```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` - -|画像 |タグ |サイズ |説明 | +| Image | Tag | Size | Description | | ------------------------ | -------- | ------ | --------------------- | -| `diegosouzapw/オムニルート` | `最新` | ~250MB |最新の安定版リリース | -| `diegosouzapw/オムニルート` | `1.0.3` | ~250MB |現在のバージョン |--- +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | + +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**新規!**OmniRoute は、Windows、macOS、Linux の**ネイティブ デスクトップ アプリケーション**として利用できるようになりました。 +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -OmniRoute をスタンドアロンのデスクトップ アプリとして実行します。ローカル モデルにはターミナル、ブラウザ、インターネットは必要ありません。 Electron ベースのアプリには次のものが含まれます。 +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**ネイティブ ウィンドウ**— システム トレイと統合された専用アプリ ウィンドウ -- 🔄**自動開始**— システムログイン時に OmniRoute を起動します -- 🔔**ネイティブ通知**— クォータの枯渇またはプロバイダーの問題に関するアラートを受け取ります -- ⚡**ワンクリック インストール**— NSIS (Windows)、DMG (macOS)、AppImage (Linux) -- 🌐**オフライン モード**— バンドルされたサーバーを使用して完全にオフラインで動作します### クイックスタート +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### クイックスタート ```bash # Development mode @@ -981,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -最小化すると、OmniRoute はシステム トレイに常駐し、簡単なアクションが実行されます。 +When minimized, OmniRoute lives in your system tray with quick actions: -- ダッシュボードを開く -- サーバーポートの変更 -- アプリケーションを終了する +- Open dashboard +- Change server port +- Quit application -📖 完全なドキュメント: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| 階層 | プロバイダー | コスト | クォータのリセット | 最適な用途 | -| ------------------------- | --------------------------------- | ------------------------------- | ------------------- | --------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 サブスクリプション** | クロード・コード (プロ) | $20/月 | 5 時間 + 毎週 | すでに購読済み | -| | コーデックス (プラス/プロ) | $20-200/月 | 5 時間 + 毎週 | OpenAI ユーザー | -| | ジェミニ CLI | **無料** | 180K/月 + 1K/日 | みんな! | -| | GitHub コパイロット | $10-19/月 | 月刊 | GitHub ユーザー | -| **🔑 API キー** | NVIDIA NIM | **無料**(開発は永久に) | ~40 RPM | 70 以上のオープン モデル | -| | 大脳 | **FREE**(1M tok/day) | 60K TPM / 30 RPM | 世界最速 | -| | グロク | **無料**(30 RPM) | 14.4K RPD | 超高速ラマ/ジェマ | -| | ディープシーク V3.2 | 100 万あたり $0.27/$1.10 | なし | 最高の価格/品質の理由 | -| | xAI Grok-4 高速 | **100 万あたり $0.20/$0.50**🆕 | なし | 最速 + ツール呼び出し、超低コスト | -| | xAI Grok-4 (標準) | 100 万あたり $0.20/$1.50 🆕 | なし | xAI の推論フラッグシップ | -| | ミストラル | 無料トライアル + 有料 | レート制限 | ヨーロッパのAI | -| | オープンルーター | 従量課金制 | なし | 合計 100 以上のモデル | -| **💰安い** | GLM-5 (Z.AI経由) 🆕 | $0.5/100万 | 毎日午前 10 時 | 128K 出力、最新のフラッグシップ | -| | GLM-4.7 | $0.6/100万 | 毎日午前 10 時 | 予算のバックアップ | -| | ミニマックス M2.5 🆕 | 100 万ドルあたり 0.3 ドルの入力 | 5時間ローリング | 推論 + エージェント タスク | -| | ミニマックス M2.1 | $0.2/100万 | 5時間ローリング | 最も安いオプション | -| | キミ K2.5 (ムーンショット API) 🆕 | 従量課金制 | None | Moonshot API への直接アクセス | -| | キミ K2 | 月額 9 ドルのフラット | 1,000 万トークン/月 | 予測可能なコスト | -| **🆓 無料** | コーダー | **$0** | 無制限 | 5モデル無制限 | -| | クウェン | **$0** | 無制限 | 4モデル無制限 | -| | キロ | **$0** | 無制限 | クロード・ソネット/俳句 (AWS ビルダー) | -| | LongCat Flash-Lite 🆕 | **$0**(5,000 万トーク/日 🔥) | 1 RPS | 地球上で最大の無料割り当て | -| | 受粉 AI 🆕 | **$0**(キーは必要ありません) | 1 リクエスト/15 秒 | GPT-5、クロード、ディープシーク、ラマ 4 | -| | Cloudflare ワーカー AI 🆕 | **$0**(10,000 ニューロン/日) | ~150 回/日 | 50 を超えるモデル、グローバル エッジ | -| | スケールウェイ AI 🆕 | **$0**(合計 100 万トークン) | レート制限 | EU/GDPR、Qwen3 235B、ラマ 70B | > 🆕**新しいモデルの追加 (2026 年 3 月):**$0.20/$0.50/M の Grok-4 Fast ファミリ (1143 ミリ秒でベンチマーク - Gemini 2.5 フラッシュより 30% 高速)、128K 出力の Z.AI 経由の GLM-5、MiniMax M2.5 推論、DeepSeek V3.2 の最新価格、Moonshot direct API 経由の Kim K2.5。 | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 $0 コンボ スタック — 完全な無料セットアップ:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**コストゼロ。コーディングを決してやめません。**これを 1 つの OmniRoute コンボとして設定すると、すべてのフォールバックが自動的に行われ、手動で切り替える必要はありません。--- +--- --- ## 🆓 Free Models — What You Actually Get -> 以下のすべてのモデルは**完全に無料で、クレジット カードは必要ありません**。 OmniRoute は、1 つのクォータがなくなると、それらの間で自動ルーティングを行います。すべてを組み合わせて、破ることのできない $0 コンボを実現します。### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -|モデル |プレフィックス |制限 |レート制限 | +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | --------------------- | -| `クロード・ソネット-4.5` | `kr/` |**無制限**| 1 日あたりの上限は報告されていません | -| `クロード俳句-4.5` | `kr/` |**無制限**| 1 日あたりの上限は報告されていません | -| `クロード作品-4.6` | `kr/` |**無制限**| Kiro 経由の最新作品 |### 🟢 QODER MODELS (Free PAT via qodercli) +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -|モデル |プレフィックス |制限 |レート制限 | +### 🟢 QODER MODELS (Free PAT via qodercli) + +| Model | Prefix | Limit | Rate Limit | | ------------------ | ------ | ------------- | --------------- | -| `kimi-k2-thing` | `if/` |**無制限**|報告された上限はありません | -| `qwen3-coder-plus` | `if/` |**無制限**|報告された上限はありません | -| `ディープシーク-r1` | `if/` |**無制限**|報告された上限はありません | -| `ミニマックス-m2.1` | `if/` |**無制限**|報告された上限はありません | -| `kimi-k2` | `if/` |**無制限**|報告された上限はありません | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -> 推奨される接続方法:**パーソナル アクセス トークン + `qodercli`**。 Browser OAuth is -> 実験的であり、`QODER_OAUTH_*` 環境変数が設定されていない限り、デフォルトで無効になっています。### 🟡 QWEN MODELS (Device Code Auth) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -|モデル |プレフィックス |制限 |レート制限 | +### 🟡 QWEN MODELS (Device Code Auth) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | ------------------- | -| `qwen3-coder-plus` | `qw/` |**無制限**|報告された上限はありません | -| `qwen3-コーダー-フラッシュ` | `qw/` |**無制限**|報告された上限はありません | -| `qwen3-coder-next` | `qw/` |**無制限**|報告された上限はありません | -| `ビジョンモデル` | `qw/` |**無制限**|マルチモーダル (画像) |### 🟣 GEMINI CLI (Google OAuth) +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -|モデル |プレフィックス |制限 |レート制限 | -| ------------------------ | ------ | ------------------------- | ------------- | -| `gemini-3-フラッシュ-プレビュー` | `gc/` |**180,000 トーク/月**+ 1,000 トーク/日 |毎月のリセット | -| `ジェミニ-2.5-プロ` | `gc/` | 180K/月 (共有プール) |高品質 |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +### 🟣 GEMINI CLI (Google OAuth) -|階層 | 1 日あたりの制限 |レート制限 |メモ | -| ---------- | ------------ | ----------- | -------------------------------------------------------- | -|無料 (開発) |トークンキャップなし |**~40 RPM**| 70以上のモデル; 2025 年半ばに純粋なレート制限に移行 | +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -人気の無料モデル: `moonshotai/kimi-k2.5` (Kimi K2.5)、`z-ai/glm4.7` (GLM 4.7)、`deepseek-ai/deepseek-v3.2` (DeepSeek V3.2)、`nvidia/llama-3.3-70b-instruct`、`deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) -|階層 | 1 日あたりの制限 |レート制限 |メモ | -| ---- | ----------------- | ---------------- | ------------------------------------------ | -|無料 |**100 万トークン/日**| 60K TPM / 30 RPM |世界最速の LLM 推論。毎日リセット | +| Tier | Daily Limit | Rate Limit | Notes | +| ---------- | ------------ | ----------- | ------------------------------------------------------ | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -無料で入手可能: `llama-3.3-70b`、`llama-3.1-8b`、`deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -|階層 | 1 日あたりの制限 |レート制限 |メモ | -| ---- | ------------- | ---------------- | -------------------------------------- | -|無料 |**14.4K RPD**|モデルごとに 30 RPM |クレジットカードはありません。 429 が制限、課金されない | +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -無料で入手可能: `llama-3.3-70b-versatile`、`gemma2-9b-it`、`mixtral-8x7b`、`whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -|モデル |プレフィックス |毎日の無料割り当て |メモ | +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` + +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ------------- | ---------------- | ----------------------------------------- | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | + +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` + +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 + +| Model | Prefix | Daily Free Quota | Notes | | ----------------------------- | ------ | ----------------- | ----------------------- | -| `LongCat-Flash-Lite` | `lc/` |**5,000 万トークン**💥 |史上最大の無料割り当て | -| `LongCat-Flash-Chat` | `lc/` | 500K トークン |マルチターンチャット | -| `LongCat-Flash-Thinking` | `lc/` | 500K トークン |推論/CoT | -| `LongCat-Flash-Thinking-2601` | `lc/` | 500K トークン | 2026年1月版 | -| `LongCat-Flash-Omni-2603` | `lc/` | 500K トークン |マルチモーダル | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | -> パブリックベータ版の間は完全に無料です。 [longcat.chat](https://longcat.chat) にメールまたは電話で登録してください。毎日 00:00 UTC にリセットされます。### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. -|モデル |プレフィックス |レート制限 |背後のプロバイダー | +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| `オープンナイ` | `ポール/` | 1 リクエスト/15 秒 | GPT-5 | -|クロード`ポール/` | 1 リクエスト/15 秒 |人間のクロード | -|ジェミニ`ポール/` | 1 リクエスト/15 秒 | Google ジェミニ | -| `ディープシーク` | `ポール/` | 1 リクエスト/15 秒 |ディープシーク V3 | -|ラマ`ポール/` | 1 リクエスト/15 秒 |メタラマ 4 スカウト | -| `ミストラル` | `ポール/` | 1 リクエスト/15 秒 |ミストラルAI | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**摩擦ゼロ:**サインアップも API キーも必要ありません。空のキーフィールドを使用して Pollinations プロバイダーを追加すると、すぐに機能します。### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -|階層 |デイリーニューロン |同等の使用法 |メモ | -| ---- | ------------- | -------------------------------------- | ----------------------- | -|無料 |**10,000**| ~150 LLM 対応 / 500 秒オーディオ / 15K 埋め込み |グローバル エッジ、50 以上のモデル | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 -人気のある無料モデル: `@cf/meta/llama-3.3-70b-instruct`、`@cf/google/gemma-3-12b-it`、`@cf/openai/whisper-large-v3-turbo` (無料音声!)、`@cf/qwen/qwen2.5-coder-15b-instruct` +| Tier | Daily Neurons | Equivalent Usage | Notes | +| ---- | ------------- | --------------------------------------- | ----------------------- | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -> [dash.cloudflare.com](https://dash.cloudflare.com) からの API トークン + アカウント ID が必要です。アカウント ID をプロバイダー設定に保存します。### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -|階層 |無料割り当て |場所 |メモ | +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. + +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | | ---- | ------------- | ------------ | ----------------------------------- | -|無料 |**100 万トークン**| 🇫🇷 パリ、EU |制限内ではクレジットカードは必要ありません | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | -無料で入手可能: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!)、`llama-3.1-70b-instruct`、`mistral-small-3.2-24b-instruct-2506`、`deepseek-v3-0324` +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` -> EU/GDPR に準拠。 [console.scaleway.com](https://console.scaleway.com) で API キーを取得します。 +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). ->**💡 究極の無料スタック (11 プロバイダー、永久に $0):** +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > ->「」 -> Kiro (kr/) → クロード・ソネット/Haiku UNLIMITED -> Qoder (if/) → kimi-k2- Thinking、qwen3-coder-plus、deepseek-r1 UNLIMITED -> LongCat Lite (lc/) → LongCat-Flash-Lite — 5,000 万トークン/日 🔥 -> 受粉 (pol/) → GPT-5、Claude、DeepSeek、Llama 4 — キーは必要ありません -> Qwen (qw/) → qwen3-coder モデル無制限 -> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 リクエスト/日無料 -> Cloudflare AI (cf/) → 50 以上のモデル — 10,000 ニューロン/日 -> Scaleway (scw/) → Qwen3 235B、Llama 70B — 100 万の無料トークン (EU) -> Groq (groq/) → ラマ/ジェマ — 14.4K 要求/日の超高速 -> NVIDIA NIM (nvidia/) → 70 以上のオープン モデル — 永久に 40 RPM -> Cerebras (cerebras/) → ラマ/クウェン世界最速 — 100 万トーク/日 ->「」## 🎙️ Free Transcription Combo +> ``` +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> あらゆるオーディオ/ビデオを**$0**で文字起こし — Deepgram は 200 ドルを無料、AssemblyAI は 50 ドルのフォールバック、無制限の緊急バックアップとして Groq Whisper を提供します。 +## 🎙️ Free Transcription Combo -|プロバイダー |無料クレジット |ベストモデル |レート制限 | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. + +| Provider | Free Credits | Best Model | Rate Limit | | ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | -| 🟢**ディープグラム**|**200 ドル無料**(サインアップ) | `nova-3` — 最高の精度、30 以上の言語 |無料クレジットには RPM 制限なし | -| 🔵**AssemblyAI**|**50 ドル無料**(サインアップ) | 「universal-3-pro」 — 章、センチメント、PII |無料クレジットには RPM 制限なし | -| 🔴**グロク**|**永久無料**| `whisper-large-v3` — OpenAI ウィスパー | 30 RPM (レート制限) | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | -**「/dashboard/combos」内の推奨コンボ:**``` +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -次に、「/dashboard/media」→「**文字起こし**」タブで、オーディオまたはビデオ ファイルをアップロードし、コンボ エンドポイントを選択し、サポートされている形式で文字起こしを取得します。## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 は、単なるリレー プロキシではなく、運用プラットフォームとして構築されています。### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| 特集 | 何をするのか | -| -------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| ⚡**Grok-4 高速ファミリー** | xAI モデルは $0.20/$0.50/M です — ベンチマークでは 1143 ミリ秒 (Gemini 2.5 フラッシュより 30% 高速) | -| 🧠**GLM-5 (Z.AI 経由)** | 128K 出力コンテキスト、0.5/100 万ドル — GLM ファミリの最新フラッグシップ | -| 🔮**ミニマックス M2.5** | 推論 + エージェント タスクを $0.30/1M で – M2.1 からの大幅なアップグレード | -| 🎯**モデルごとのツール呼び出しフラグ** | レジストリ内のモデルごとの `toolCalling: true/false` — AutoCombo はツール非対応モデルをスキップします。 | -| 🌍**多言語の意図の検出** | オートコンボ スコアリングにおける PT/ZH/ES/AR キーワード — 英語以外のコンテンツに対するより適切なモデル選択 | -| 📊**ベンチマーク主導のフォールバック** | ライブリクエストからの実際の p95 レイテンシがコンボスコアリングにフィード — AutoCombo は実際のデータから学習 | -| 🔁**重複排除をリクエスト** | コンテンツ ハッシュ ベースの重複排除ウィンドウ - マルチエージェントで安全、重複請求を防止 | -| 🔌**プラグイン可能なルーター戦略** | 拡張可能な「RouterStrategy」インターフェイス — カスタム ルーティング ロジックをプラグインとして追加します。### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| 特集 | 何をするのか | -| --------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**モデルの遊び場** | 任意のモデルを直接テストするためのダッシュボード ページ — プロバイダー/モデル/エンドポイント セレクター、モナコ エディター、ストリーミング、中止、タイミング | -| 🔏**CLI 指紋照合** | ネイティブ CLI 署名と一致するようにプロバイダーごとにヘッダー/本文を順序付けします。[設定] > [セキュリティ] でプロバイダーごとに切り替えます。**プロキシ IP は保持されます** | -| 🤝**ACP サポート (エージェント クライアント プロトコル)** | CLI エージェント検出 (Codex、Claude、Goose、Gemini CLI、OpenClaw + その他 9 つ)、プロセス スポーナー、`/api/acp/agents` エンドポイント | -| 🤖**ACP エージェント ダッシュボード** | デバッグ › エージェント ページ — インストール ステータス、バージョン、CLI ツールのカスタム エージェント フォームを含む 14 個のエージェントのグリッド。**OpenCode**ユーザーには、利用可能なすべてのモデルですぐに使用できる構成を自動生成する [opencode.json をダウンロード] ボタンが表示されます。 | -| 🔧**カスタム モデル `apiFormat` ルーティング** | `apiFormat: "responses"` を持つカスタム モデルは、Responses API トランスレーターに正しくルーティングされるようになりました。 | -| 🏢**Codex ワークスペースの分離** | 電子メールごとに複数の Codex ワークスペース - OAuth はワークスペース ID によって接続を正しく分離します。 | -| 🔄**Electron 自動アップデート** | デスクトップ アプリがアップデートをチェックし、再起動時に自動インストール | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| 特集 | 何をするのか | -| --------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**MCP サーバー (25 ツール)** | 3 つのトランスポート経由の IDE/エージェント ツール: stdio、SSE (`/api/mcp/sse`)、ストリーミング可能な HTTP (`/api/mcp/stream`)。 18 コア + 3 メモリ + 4 スキルツール | -| 🤝**A2A サーバー (JSON-RPC + SSE)** | 同期およびストリーミング フローを使用したエージェント間のタスクの実行 | -| 🧭**統合エンドポイント ページ** | [エンドポイント プロキシ]、[MCP]、[A2A]、および [API エンドポイント] タブを備えたタブ付き管理ページ | -| 🎚️**サービスの有効化/無効化の切り替え** | 設定永続性を備えた MCP および A2A の ON/OFF スイッチ (デフォルト: OFF) | -| 🛰️**MCP ランタイム ハートビート** | 実際のプロセス ステータス (pid、稼働時間、ハートビート経過時間、トランスポート、スコープ モード) | -| 📋**MCP 監査証跡** | 成功/失敗およびキーの属性を含むフィルタリング可能な監査ログ | -| 🔐**MCP スコープの適用** | 制御されたツールアクセスのための 10 の詳細な範囲の権限 | -| 📡**A2A タスク ライフサイクル管理** | タスクのリスト/フィルター、イベント/アーティファクトの検査、実行中のタスクのキャンセル | -| 📋**エージェント カードの検出** | クライアント自動検出用の `/.well-known/agent.json` | -| 🧪**プロトコル E2E テスト ハーネス** | `test:protocols:e2e` での実際の MCP SDK + A2A クライアント フロー | -| ⚙️**運用管理** | 1 つのコントロール サーフェスからコンボを切り替え、レジリエンス プロファイルを適用し、ブレーカーをリセット | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| 特集 | 何をするのか | -| ------------------------------------------ | ----------------------------------------------------------------------------------- | ----------------------- | -| 🎯**スマート 4 層フォールバック** | 自動ルート: サブスクリプション → API キー → 格安 → 無料 | -| 📊**リアルタイムのクォータ追跡** | プロバイダーごとのライブ トークン数 + リセット カウントダウン | -| 🔄**フォーマット変換** | OpenAI ↔ Claude ↔ Gemini ↔ スキーマセーフな変換による応答 | -| 👥**マルチアカウントのサポート** | インテリジェントな選択によるプロバイダーごとの複数のアカウント | -| 🔄**自動トークン更新** | OAuth トークンは再試行によって自動的に更新されます。 | -| 🎨**カスタムコンボ** | 9 つのバランシング戦略 + フォールバック チェーン制御 | -| 🌐**ワイルドカードルーター** | `provider/*` 動的ルーティング | -| 🧠**予算管理について考える** | パススルー、自動、カスタム、および適応推論の制限 | -| 🔀**モデルのエイリアス** | 組み込み + カスタム モデルのエイリアシングと移行の安全性 | -| ⚡**背景の劣化** | 優先度の低いバックグラウンド タスクを安価なモデルにルーティングする | -| 🧪**タスク認識型スマート ルーティング** | コンテンツ タイプ (コーディング/ビジョン/分析/要約) によるモデルの自動選択 | -| 🔄**A2A エージェントのワークフロー** | ステートフルな複数ステップのエージェント実行のための決定論的 FSM オーケストレーター | -| 🔀**適応ルーティング** | トークン量とプロンプトの複雑さに基づいた動的な戦略オーバーライド | -| 🎲**プロバイダーの多様性** | シャノンのエントロピー スコアリング バランシング オートコンボ トラフィック分散 | -| 💬**システム プロンプト インジェクション** | 一貫して適用されるグローバルな動作制御 | -| 📄**レスポンス API の互換性** | Codex および高度なエージェント ワークフローの完全な `/v1/responses` サポート | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| 特集 | 何をするのか | -| ---------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**画像生成** | クラウドとローカルのバックエンドを備えた `/v1/images/generations` | -| 📐**埋め込み** | 検索および RAG パイプライン用の `/v1/embeddings` | -| 🎤**音声文字起こし** | `/v1/audio/transcriptions` — 7 プロバイダー (Deepgram Nova 3、AssemblyAI、Groq Whisper、HuggingFace、イレブンラボ、OpenAI、Azure)、自動言語検出、MP4/MP3/WAV サポート | -| 🔊**テキスト読み上げ** | `/v1/audio/speech` — 10 プロバイダー (Celebrities、OpenAI、Deepgram、Cartesia、PlayHT、HuggingFace、Nvidia NIM、I​​nworld、Coqui、Tortoise) と正しいエラー メッセージ | -| 🎬**ビデオ生成** | `/v1/videos/generations` (ComfyUI + SD WebUI ワークフロー) | -| 🎵**音楽生成** | `/v1/music/generations` (ComfyUI ワークフロー) | -| 🛡️**モデレーション** | `/v1/moderations` の安全性チェック | -| 🔀**再ランキング** | 関連性スコアリング用の `/v1/rerank` | -| 🔍**ウェブ検索**🆕 | `/v1/search` — 5 プロバイダー (Serper、Brave、Perplexity、Exa、Tavily)、6,500 以上無料/月、自動フェイルオーバー、キャッシュ | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| 特集 | 何をするのか | -| ---------------------------------------------- | --------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**サーキットブレーカー** | 閾値制御によるモデルごとのトリップ/リカバリ | -| 🎯**エンドポイント対応モデル** | カスタム モデルは、サポートされるエンドポイント + API 形式を宣言します。 | -| 🛡️**対雷鳴の群れ** | 再試行/レートイベントにおけるミューテックス + セマフォ保護 | -| 🧠**セマンティック + 署名キャッシュ** | 2 つのキャッシュ層によるコスト/遅延の削減 | -| ⚡**冪等性のリクエスト** | 重複保護ウィンドウ | -| 🔒**TLS 指紋スプーフィング** | ブラウザのような TLS フィンガープリント —**ボットの検出とアカウントのフラグ付けを削減します** | -| 🔏**CLI 指紋照合** | ネイティブ CLI リクエスト署名と一致 —**プロキシ IP を保持しながら禁止リスクを軽減** | -| 🌐**IP フィルタリング** | 公開された展開のホワイトリスト/ブロックリストの制御 | -| 📊**編集可能なレート制限** | 永続性を備えた構成可能なグローバル/プロバイダーレベルの制限 | -| 📉**グレースフル デグラデーション** | コアゲートウェイの動作を保護するマルチレイヤー機能フォールバック | -| 📜**構成監査証跡** | 差分ベースの変更追跡により、単純なロールバックで運用上のドリフトを防止 | -| ⏳**プロバイダーヘルス同期** | プロアクティブなトークン有効期限監視により、認証が失敗する前にアラートをトリガー | -| 🚪**禁止されたアカウントを自動的に無効にする** | 永久にブロックされたトークン アカウントを自動的に封印する運用上のサーキット ブレーカー | -| 🔑**API キー管理 + スコーピング** | 安全なキーの発行/ローテーションとモデル/プロバイダーの制御 | -| 👁️**スコープ指定された API キーの公開**🆕 | `ALLOW_API_KEY_REVEAL` による API キーのオプトイン回復 | -| 🛡️**保護された `/models`** | モデル カタログのオプションの認証ゲートとプロバイダーの非表示 | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| 特集 | 何をするのか | -| ------------------------------------ | --------------------------------------------------------------------- | ---------------------------- | -| 📝**リクエスト + プロキシ ログ** | 完全なリクエスト/レスポンスとプロキシ ログ | -| 📉**ストリーミングされた詳細ログ**🆕 | SSE ペイロード ストリームを UI にきれいに再構築します。 | -| 📋**統合ログ ダッシュボード** | リクエスト、プロキシ、監査、およびコンソールのビューを 1 ページに表示 | -| 🔍**テレメトリのリクエスト** | p50/p95/p99 レイテンシーとリクエストのトレース | -| 🏥**健康ダッシュボード** | 稼働時間、ブレーカー状態、ロックアウト、キャッシュ統計 | -| 💰**コスト追跡** | 予算管理とモデルごとの価格の可視性 | -| 📈**分析の視覚化** | モデル/プロバイダーの使用状況に関する洞察と傾向ビュー | -| 🧪**評価フレームワーク** | 構成可能な一致戦略を使用したゴールデン セット テスト | -| 📡**ライブ診断**🆕 | 正確なコンボ ライブ テストのためのセマンティック キャッシュ バイパス | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| 特集 | 何をするのか | -| ------------------------------------- | ------------------------------------------------------------------------------------------------ | ------------------------------ | -| 🌐**どこにでも展開** | Localhost、VPS、Docker、クラウド環境 | -| 🚇**クラウドフレア トンネル**🆕 | ダッシュボードからのワンクリックのクイック トンネル統合 | -| 🔑**API キー モデルのフィルタリング** | 割り当てられたベアラー コンテキスト ロールによってフィルタリングされたネイティブ /v1/models 応答 | -| ⚡**スマート キャッシュ バイパス** | 構成可能な TTL ヒューリスティックと強制再フェッチ制御 | -| 🔄**バックアップ/復元** | エクスポート/インポートと災害復旧フロー | -| 🧙**オンボーディング ウィザード** | 初回実行のガイド付きセットアップ | -| 🔧**CLI ツール ダッシュボード** | 人気のコーディング ツールをワンクリックでセットアップ | -| 🎮**モデルの遊び場** | ダッシュボードから任意のプロバイダー/モデル/エンドポイントをテストします。 | -| 🔏**CLI 指紋切り替え** | [設定] > [セキュリティ] | でのプロバイダーごとの指紋照合 | -| 🌐**i18n (30 言語)** | RTL をカバーする完全なダッシュボード + ドキュメント言語サポート | -| 🧹**すべてのモデルをクリア** | ワンクリックでプロバイダー詳細のモデルリストをクリア | -| 👁️**サイドバー コントロール**🆕 | 外観設定からコンポーネントと統合を非表示にする | -| 📋**問題テンプレート** | バグと機能用の標準化された GitHub テンプレート | -| 📂**カスタム データ ディレクトリ** | 格納場所の `DATA_DIR` オーバーライド | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1294,105 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -クォータ、レート、またはヘルスに障害が発生した場合、OmniRoute は手動で切り替えることなく、次の候補に自動的に移動します。#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A は UI とドキュメントで検出可能です (非表示ではありません) -- プロトコル ステータス API はライブ運用データを公開します (`/api/mcp/*`、`/api/a2a/*`) -- ダッシュボードには 2 日目の操作のアクション (コンボの切り替え、ブレーカーのリセット、タスクのキャンセル) が含まれています#### Translator + validation workflow +#### Protocol management that is visible and operable -翻訳領域には次のものが含まれます。 +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**プレイグラウンド**: 変換チェックをリクエストします -**チャット テスター**: 完全なリクエスト/レスポンスの往復 -**テストベンチ**: 1 回の実行で複数のケース -**ライブ モニター**: リアルタイムの交通状況ビュー +#### Translator + validation workflow -さらに、「npm run test:protocols:e2e」による実際のクライアントでのプロトコル検証。 +The Translator area includes: -> 📖**[MCP サーバー README](open-sse/mcp-server/README.md)**— ツール リファレンス、IDE 構成、およびクライアントの例 +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A サーバー README](src/lib/a2a/README.md)**— スキル、JSON-RPC メソッド、ストリーミング、タスクのライフサイクル## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute には、ゴールデン セットに対して LLM 応答品質をテストするための評価フレームワークが組み込まれています。ダッシュボードの**Analytics → Evals**からアクセスします。### Built-in Golden Set +## 🧪 Evaluations (Evals) -プリロードされた「OmniRoute Golden Set」には、次のテスト ケースが含まれています。 +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- 挨拶、数学、地理、コード生成 -- JSON形式への準拠、翻訳、マークダウン生成 -- 安全拒否(有害なコンテンツ)、カウント、ブール論理### Evaluation Strategies +### Built-in Golden Set -| 戦略 | 説明 | 例 | -| ---------- | ------------------------------------------------------------------------------------------ | ----------- | -| '正確' | 出力は正確に一致する必要があります | `"4"` | -| `を含む` | 出力には部分文字列が含まれている必要があります (大文字と小文字は区別されません)。 「パリ」 | -| `正規表現` | 出力は正規表現パターンと一致する必要があります | `"1.*2.*3"` | -| `カスタム` | カスタム JS 関数は true/false を返します。 `(出力) => 出力.長さ > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) -<詳細> +
+🧩 MCP Setup (Model Context Protocol) -🧩 MCP セットアップ (モデル コンテキスト プロトコル) +Start MCP transport in stdio mode: -MCP トランスポートを標準入出力モードで開始します。```bash +```bash omniroute --mcp +``` -```` +Recommended validation flow: -推奨される検証フロー: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. MCP クライアントを標準入出力経由で接続します。 -2. 「omniroute_get_health」を実行します。 -3. 「omniroute_list_combos」を実行します。 -4. 「/dashboard/mcp」を開いて、ハートビート、アクティビティ、および監査を確認します。 +Useful APIs for automation: -自動化に役立つ API: - -- `/api/mcp/statusを取得` -- `/api/mcp/tools` を取得します +- `GET /api/mcp/status` +- `GET /api/mcp/tools` - `GET /api/mcp/audit` -- `/api/mcp/audit/stats` を取得します
+- `GET /api/mcp/audit/stats` -<詳細> -🤝 A2A セットアップ (Agent2Agent) + -エージェントを検出します。```bash +
+🤝 A2A Setup (Agent2Agent) + +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -タスクを送信します。```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` - -ライフサイクルを管理する: +Manage lifecycle: - `GET /api/a2a/status` -- `/api/a2a/tasks`を取得します +- `GET /api/a2a/tasks` - `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -操作UI: +Operational UI: -- タスク/状態/ストリームの可観測性とスモークアクション用の「/dashboard/a2a」
+- `/dashboard/a2a` for task/state/stream observability and smoke actions -<詳細> -🧪 エンドツーエンドのプロトコル検証 + -実際のクライアントを使用して両方のプロトコルを検証します。```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -これにより以下が検証されます。 +This verifies: -- MCP SDK クライアントの接続/リスト/呼び出し -- A2A 検出/送信/ストリーム/取得/キャンセル -- MCP監査およびA2Aタスク管理APIのデータをクロスチェックします
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs -<詳細> + -💳 サブスクリプション プロバイダー### Claude Code (Pro/Max) +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1405,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**プロのヒント:**複雑なタスクには Opus を使用し、速度を求める場合は Sonnet を使用します。 OmniRoute はモデルごとの割り当てを追跡します。### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1419,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -各 Codex アカウントには、「ダッシュボード -> プロバイダー」でポリシーの切り替えができるようになりました。 +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (オン/オフ): 5 時間のウィンドウしきい値ポリシーを適用します。 -- 「毎週」 (オン/オフ): 毎週のウィンドウしきい値ポリシーを適用します。 -- しきい値の動作: 有効なウィンドウの使用率が 90% 以上に達すると、そのアカウントはスキップされます。 -- ローテーション動作: OmniRoute は、次の対象となる Codex アカウントに自動的にルーティングします。 -- リセット動作: プロバイダーの「resetAt」時間が経過すると、アカウントは自動的に再び適格になります。 +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -シナリオ: +Scenarios: -- 「5 時間 ON」 + 「毎週 ON」: いずれかのウィンドウがしきい値に達すると、アカウントはスキップされます。 -- 「5 時間オフ」 + 「毎週オン」: 毎週の使用のみアカウントをブロックできます。 -- 「5時間オン」 + 「毎週オフ」: 5時間の使用のみでアカウントをブロックできます。 -- `resetAt` が渡されました: アカウントは自動的にローテーションを再開します (手動での再有効化はありません)。### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1444,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**ベストバリュー:**膨大な無料枠!有料レベルの前にこれを使用してください。### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1459,73 +1662,91 @@ Models:
-<詳細> +
+🔑 API Key Providers -🔑 API キー プロバイダー### NVIDIA NIM (FREE developer access — 70+ models) +### NVIDIA NIM (FREE developer access — 70+ models) -1. サインアップ: [build.nvidia.com](https://build.nvidia.com) -2. 無料の API キーを取得します (1000 推論クレジットが含まれます) -3. ダッシュボード → プロバイダーの追加 → NVIDIA NIM: - - API キー: `nvapi-your-key` +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**モデル:**`nvidia/llama-3.3-70b-instruct`、`nvidia/mistral-7b-instruct`、および 50 以上 +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -**プロのヒント:**OpenAI 互換 API — OmniRoute のフォーマット変換とシームレスに連携します。### DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -1. サインアップ: [platform.deepseek.com](https://platform.deepseek.com) -2. APIキーを取得する -3. ダッシュボード → プロバイダーの追加 → DeepSeek +### DeepSeek -**モデル:**`ディープシーク/ディープシークチャット`、`ディープシーク/ディープシークコーダー`### Groq (Free Tier Available!) +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -1. サインアップ: [console.groq.com](https://console.groq.com) -2. API キーを取得します (無料利用枠を含む) -3. ダッシュボード → プロバイダーの追加 → Groq +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**モデル:**`groq/llama-3.3-70b`、`groq/mixtral-8x7b` +### Groq (Free Tier Available!) -**プロのヒント:**超高速推論 — リアルタイム コーディングに最適です。### OpenRouter (100+ Models) +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -1. サインアップ: [openrouter.ai](https://openrouter.ai) -2. APIキーを取得する -3. ダッシュボード → プロバイダーの追加 → OpenRouter +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**モデル:**単一の API キーを通じて、すべての主要プロバイダーの 100 以上のモデルにアクセスします。 +**Pro Tip:** Ultra-fast inference — best for real-time coding! -**ダッシュボードの動作:**OpenRouter モデルは**利用可能なモデル**から管理されます。手動追加、インポート、自動同期はすべて同じリストを更新します。
+### OpenRouter (100+ Models) -<詳細> +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -💰 格安プロバイダー (バックアップ)### GLM-4.7 (Daily reset, $0.6/1M) +**Models:** Access 100+ models from all major providers through a single API key. -1. サインアップ: [Zhipu AI](https://open.bigmodel.cn/) 2.コーディングプランからAPIキーを取得 -2. ダッシュボード → API キーの追加: - - プロバイダー: `glm` - - API キー: `your-key` +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -**使用:**`glm/glm-4.7` + -**プロのヒント:**コーディング プランでは、1/7 のコストで 3 倍の割り当てを提供します。毎日午前 10 時にリセットされます。### MiniMax M2.1 (5h reset, $0.20/1M) +
+💰 Cheap Providers (Backup) -1. サインアップ:[MiniMax](https://www.minimax.io/) -2. APIキーを取得する -3. ダッシュボード → APIキーの追加 +### GLM-4.7 (Daily reset, $0.6/1M) -**使用方法:**`minimax/MiniMax-M2.1` +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**プロのヒント:**長いコンテキスト (100 万トークン) の最も安価なオプション!### Kimi K2 ($9/month flat) +**Use:** `glm/glm-4.7` -1. 購読:[ムーンショット AI](https://platform.moonshot.ai/) -2. APIキーを取得する -3. ダッシュボード → APIキーの追加 +**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. -**使用方法:**`kimi/kimi-latest` +### MiniMax M2.1 (5h reset, $0.20/1M) -**プロのヒント:**1,000 万トークンの固定 $9/月 = 0.90 ドル/100 万の実効コスト!
+1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key -<詳細> +**Use:** `minimax/MiniMax-M2.1` -🆓 無料プロバイダー (緊急バックアップ)### Qoder (5 FREE models via OAuth) +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1566,9 +1787,10 @@ Models:
-<詳細> +
+🎨 Create Combos -🎨 コンボを作成する### Example 1: Maximize Subscription → Cheap Backup +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1596,9 +1818,10 @@ Cost: $0 forever!
-<詳細> +
+🔧 CLI Integration -🔧 CLI の統合### Cursor IDE +### Cursor IDE ``` Settings → Models → Advanced: @@ -1609,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -ダッシュボードの**CLI ツール**ページを使用してワンクリック構成を行うか、`~/.claude/settings.json` を手動で編集します。### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1620,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**オプション 1 — ダッシュボード (推奨):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**オプション 2 — 手動:**`~/.openclaw/openclaw.json` を編集します:```json +```json { "models": { "providers": { @@ -1637,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **注:**OpenClaw はローカル OmniRoute でのみ機能します。 IPv6 解決の問題を回避するには、「localhost」の代わりに「127.0.0.1」を使用してください。### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1651,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**ステップ 1:**OmniRoute をカスタム プロバイダーとして追加します。```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**ステップ 2:**プロジェクト ルートで `opencode.json` を作成/編集します。```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1677,118 +1909,130 @@ opencode } } } -```` +``` -**ステップ 3:**OpenCode でモデルを選択します。```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**ヒント:**OmniRoute `/v1/models` エンドポイントで利用可能なモデルを `models` セクションに追加します。 OmniRoute ダッシュボードの「provider/model-id」形式を使用します。
+ --- ## トラブルシューティング -<詳細> -クリックしてトラブルシューティング ガイドを展開します +
+Click to expand troubleshooting guide -**「言語モデルがメッセージを提供しませんでした」** +**"Language model did not provide messages"** -- プロバイダー クォータが枯渇した → ダッシュボード クォータ トラッカーを確認してください -- 解決策: コンボフォールバックを使用するか、より安価なレベルに切り替える +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**レート制限** +**Rate limiting** -- サブスクリプション クォータ アウト → GLM/MiniMax へのフォールバック -- コンボを追加: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2- Thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**OAuth トークンの有効期限が切れました** +**OAuth token expired** -- OmniRouteによる自動更新 -- 問題が解決しない場合: ダッシュボード → プロバイダー → 再接続 +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**高コスト** +**High costs** -- [ダッシュボード] → [コスト] で使用状況の統計を確認します。 -- プライマリ モデルを GLM/MiniMax に切り替えます -- 重要ではないタスクには無料枠 (Gemini CLI、Qoder) を使用する +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**ダッシュボード/API ポートが間違っています** +**Dashboard/API ports are wrong** -- `PORT` は正規のベースポート (デフォルトでは API ポート) です。 -- `API_PORT` は OpenAI 互換の API リスナーのみをオーバーライドします -- `DASHBOARD_PORT` はダッシュボード/Next.js リスナーのみをオーバーライドします -- `NEXT_PUBLIC_BASE_URL` をダッシュボード/パブリック URL に設定します (OAuth コールバック用) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**クラウド同期エラー** +**Cloud sync errors** -- 「BASE_URL」が実行中のインスタンスを指していることを確認します -- 「CLOUD_URL」が予想されるクラウド エンドポイントを指していることを確認します -- `NEXT_PUBLIC_*` 値をサーバー側の値と一致させます。 +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**最初のログインが機能しない** +**First login not working** -- `.env`の`INITIAL_PASSWORD`を確認してください -- 設定されていない場合、フォールバックパスワードは「123456」です。 +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**リクエストログなし** +**No request logs** -- リクエスト アーティファクトは、リクエストごとに 1 つの JSON ファイルとして「DATA_DIR/call_logs/」に書き込まれます。 -- 詳細なステージごとのペイロードが必要な場合は、「ダッシュボード」→「ログ」→「ログのリクエスト」からパイプライン キャプチャを有効にします。 -- `logs/application/app.log` にもアプリケーション コンソール ログが必要な場合は、`APP_LOG_TO_FILE=true` を設定します。 -- 必要に応じて、「APP_LOG_MAX_FILE_SIZE」、「APP_LOG_RETENTION_DAYS」、「APP_LOG_MAX_FILES」、「CALL_LOG_MAX_ENTRIES」を調整します。 +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**OpenAI 互換プロバイダーの接続テストで「無効」と表示される** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- 多くのプロバイダーは「/models」エンドポイントを公開していません -- OmniRoute v1.0.6+ には、チャット完了によるフォールバック検証が含まれています -- ベース URL に「/v1」サフィックスが含まれていることを確認してください### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ VPS、Docker、または任意のリモート サーバーで OmniRoute を実行しているユーザーにとって重要**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -**Antigravity**プロバイダーと**Gemini CLI**プロバイダーは**Google OAuth 2.0**を使用します。 Google では、OAuth フロー内の「redirect_uri」が、アプリの Google Cloud Console に事前登録された URI の 1 つと正確に一致することを要求しています。 +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -OmniRoute にバンドルされている OAuth 資格情報は、**「localhost」に対してのみ**登録されます。リモート サーバー (例: 「https://omniroute.myserver.com」) 上の OmniRoute にアクセスすると、Google は次のような認証を拒否します。``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Google Cloud Console でサーバーの URI を使用して**OAuth 2.0 クライアント ID**を作成する必要があります。#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Google Cloud コンソールを開きます** +#### Step-by-step -[https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) に移動します。 +**1. Open Google Cloud Console** -**2.新しい OAuth 2.0 クライアント ID を作成します** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) --**「+ 資格情報の作成」**→**「OAuth クライアント ID」**をクリックします。 +**2. Create a new OAuth 2.0 Client ID** -- アプリケーションの種類:**「Web アプリケーション」** -- 名前: 任意の名前 (例:「OmniRoute Remote」) +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -**3.承認されたリダイレクト URI の追加** +**3. Add Authorized Redirect URIs** -**「承認されたリダイレクト URI」**フィールドに、次を追加します。``` +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> 「your-server.com」をサーバーのドメインまたはIPに置き換えます(必要に応じてポートを含めます、例:「http://45.33.32.156:20128/callback」)。 +**4. Save and copy the credentials** -**4.資格情報を保存してコピーします** +After creating, Google will show the **Client ID** and **Client Secret**. -作成後、Google には**クライアント ID**と**クライアント シークレット**が表示されます。 +**5. Set environment variables** -**5.環境変数を設定します** +In your `.env` (or Docker environment variables): -`.env` (または Docker 環境変数) 内:```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1797,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. OmniRoute を再起動します**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7.もう一度接続してみてください** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -ダッシュボード → プロバイダー → Antigravity (または Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google は「https://your-server.com/callback」に正しくリダイレクトされるようになりました。--- +--- #### Temporary workaround (without custom credentials) -今すぐ独自の認証情報を設定したくない場合でも、**手動 URL フロー**を使用できます。 +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute が Google 認証 URL を開きます -2. 承認後、Google は「localhost」へのリダイレクトを試行します (リモート サーバーでは失敗します)。 -3. ブラウザのアドレス バーから**完全な URL をコピー**します (ページが読み込まれない場合でも) -4. その URL を OmniRoute 接続モーダルに表示されるフィールドに貼り付けます。 -5.**「接続」**をクリックします。 +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> これは、リダイレクト ページが読み込まれたかどうかに関係なく、URL 内の認証コードが有効であるため機能します。--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. -<詳細> -🇧🇷 ポルトガル語について#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -**反重力**と**Gemini CLI**を使用して**Google OAuth 2.0**を認証します。 O Google は、「redirect_uri」を使用して、フラクソ OAuth 認証を実行せず、**exatamente**の URI を事前に管理し、Google Cloud Console でアプリケーションを実行しません。 +
+🇧🇷 Versão em Português -認証情報として OAuth は、「localhost」**に対する OmniRoute データベースの構築をサポートしません。 OmniRoute サーバー リモートへのアクセス (例: `https://omniroute.meuservidor.com`)、Google による認証:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -**OAuth 2.0 クライアント ID**では、Google Cloud Console の URI がサーバーを参照する必要があります。#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Google Cloud コンソールへのアクセス** +#### Passo a passo -アブラ: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Acesse o Google Cloud Console** -**2.新しい OAuth 2.0 クライアント ID** +Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Clique em**「+ 認証情報の作成」**→**「OAuth クライアント ID」** -- 応用情報:**「Web アプリケーション」** -- 名前: escolha qualquer nome (例: `OmniRoute Remote`) +**2. Crie um novo OAuth 2.0 Client ID** -**3.承認されたリダイレクト URI としての Adicione** +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**「承認されたリダイレクト URI」**はありません、アディシオン:``` +**3. Adicione as Authorized Redirect URIs** + +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> `seu-servidor.com` をサーバーの IP アドレスに置き換えます (必要なポータルを含む、例: `http://45.33.32.156:20128/callback`)。 +**4. Salve e copie as credenciais** -**4.コピーを認証情報として保存** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Google のほとんどの情報、**クライアント ID**、または**クライアント シークレット**を使用してください。 +**5. Configure as variáveis de ambiente** -**5.環境変数として設定** +No seu `.env` (ou nas variáveis de ambiente do Docker): -`.env` はありません (Docker の環境変数を変更します):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1876,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6.レイニシー・オ・オムニルート**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute +``` -```` +**7. Tente conectar novamente** -**7.テンテ コネクター ノヴァメンテ** +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -ダッシュボード → プロバイダー → Antigravity (Gemini CLI) → OAuth +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. -`https://seu-servidor.com/callback` の機能を確認するには、Google からアクセスしてください。--- +--- #### Workaround temporário (sem configurar credenciais próprias) -**URL のマニュアル**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. OmniRoute は Google の自動 URL を提供します -2. `localhost` による Google の自動リダイレクト (サーバー リモートの呼び出し) -3.**URL をコピーしてください**ブラウザを使用してブラウザを開きます (カレーグのページを表示します) -4. コール エッサ URL は、OmniRoute の接続モーダルなしです。 -5. クリーク**「接続」** +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) +4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute +5. Clique em **"Connect"** -> 自動回避策の機能は URL から独立してリダイレクトされます。
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1914,64 +2171,73 @@ docker restart omniroute ## 🛠️ Tech Stack -<詳細> -クリックして技術スタックの詳細を展開します +
+Click to expand tech stack details --**ランタイム**: Node.js 18–22 LTS (⚠️ Node.js 24+ は**サポートされていません**— `better-sqlite3` ネイティブ バイナリは互換性がありません) --**言語**: TypeScript 5.9 - `src/` と `open-sse/` で**100% TypeScript**(v2.0 以降のコア モジュールには `any` はありません) --**フレームワーク**: Next.js 16 + React 19 + Tailwind CSS 4 --**データベース**: LowDB (JSON) + SQLite (ドメイン状態 + プロキシ ログ + MCP 監査 + ルーティング決定) --**スキーマ**: Zod (MCP ツール I/O 検証、API コントラクト) --**プロトコル**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**ストリーミング**: サーバー送信イベント (SSE) --**認証**: OAuth 2.0 (PKCE) + JWT + API キー + MCP スコープの認証 --**テスト**: Node.js テスト ランナー + Vitest (単体、統合、E2E を含む 900 以上のテスト) --**CI/CD**: GitHub アクション (自動 npm パブリッシュ + リリース時の Docker Hub) --**ウェブサイト**: [omniroute.online](https://omniroute.online) --**パッケージ**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**回復力**: サーキット ブレーカー、エクスポネンシャル バックオフ、アンチサンダー ハード、TLS スプーフィング、自動コンボ自己修復
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## ドキュメント -|ドキュメント |説明 | -| ------------------------------------------------ | --------------------------------------------------- | -| [ユーザーガイド](docs/USER_GUIDE.md) |プロバイダー、コンボ、CLI 統合、展開 | -| [API リファレンス](docs/API_REFERENCE.md) |すべてのエンドポイントと例 | -| [MCP サーバー](open-sse/mcp-server/README.md) | 16 MCP ツール、IDE 構成、Python/TS/Go クライアント | -| [A2A サーバー](src/lib/a2a/README.md) | JSON-RPC 2.0 プロトコル、スキル、ストリーミング、タスク管理 | -| [オートコンボ エンジン](docs/auto-combo.md) | 6 要素スコアリング、モード パック、自己修復 | -| [トラブルシューティング](docs/TROUBLESHOOTING.md) |よくある問題と解決策 | -| [アーキテクチャ](docs/ARCHITECTURE.md) |システム アーキテクチャと内部構造 | -| [寄稿](CONTRIBUTING.md) |開発セットアップとガイドライン | -| [OpenAPI 仕様](docs/openapi.yaml) | OpenAPI 3.0 仕様 | -| [セキュリティポリシー](SECURITY.md) |脆弱性の報告とセキュリティの実践 | -| [VM の展開](docs/VM_DEPLOYMENT_GUIDE.md) |完全ガイド: VM + nginx + Cloudflare セットアップ | -| [機能ギャラリー](docs/FEATURES.md) |スクリーンショットを含むビジュアル ダッシュボード ツアー | -| [リリースチェックリスト](docs/RELEASE_CHECKLIST.md) |リリース前の検証手順 |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute には、複数の開発フェーズにわたって**210 以上の機能が計画されています**。主要な領域は次のとおりです。 +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -|カテゴリー |計画されている機能 |ハイライト | +| Category | Planned Features | Highlights | | ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | -| 🧠**ルーティングとインテリジェンス**| 25+ |最低レイテンシのルーティング、タグベースのルーティング、クォータ プリフライト、P2C アカウントの選択 | -| 🔒**セキュリティとコンプライアンス**| 20歳以上 | SSRF の強化、資格情報のクローキング、エンドポイントごとのレート制限、管理キーのスコーピング | -| 📊**可観測性**| 15 歳以上 | OpenTelemetry 統合、リアルタイム クォータ監視、モデルごとのコスト追跡 | -| 🔄**プロバイダーの統合**| 20歳以上 |動的モデル レジストリ、プロバイダーのクールダウン、マルチアカウント Codex、Copilot クォータ解析 | -| ⚡**パフォーマンス**| 15 歳以上 |デュアル キャッシュ レイヤー、プロンプト キャッシュ、応答キャッシュ、ストリーミング キープアライブ、バッチ API | -| 🌐**生態系**| 10+ | WebSocket API、構成ホットリロード、分散構成ストア、商用モード |### 🔜 Coming Soon +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**OpenCode Integration**— OpenCode AI コーディング IDE のネイティブ プロバイダー サポート -- 🔗**TRAE 統合**— TRAE AI 開発フレームワークの完全サポート -- 📦**バッチ API**— 一括リクエストの非同期バッチ処理 -- 🎯**タグベースのルーティング**— カスタムタグとメタデータに基づいてリクエストをルーティングします -- 💰**最低コスト戦略**— 利用可能な最も安価なプロバイダーを自動的に選択します +### 🔜 Coming Soon -> 📝 全機能仕様は [`docs/new-features/`](docs/new-features/) で入手可能 (217 の詳細仕様)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1979,18 +2245,20 @@ OmniRoute には、複数の開発フェーズにわたって**210 以上の機 ### How to Contribute -1. リポジトリをフォークする -2. 機能ブランチを作成します (`git checkout -b feature/amazing-feature`) -3. 変更をコミットします (`git commit -m '素晴らしい機能を追加'`) -4. ブランチにプッシュします (`git Push Origin feature/amazing-feature`) -5. プルリクエストを開く +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -詳細なガイドラインについては、[CONTRIBUTING.md](CONTRIBUTING.md) を参照してください。### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -2002,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -このフォークのきっかけとなった元のプロジェクトである**[decolua](https://github.com/decolua)**による**[9router](https://github.com/decolua/9router)**に感謝します。 OmniRoute は、追加機能、マルチモーダル API、完全な TypeScript の書き換えを備えた素晴らしい基盤の上に構築されています。 +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— この JavaScript 移植のきっかけとなったオリジナルの Go 実装に感謝します。--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## ライセンス -MIT ライセンス - 詳細については、[LICENSE](LICENSE) を参照してください。--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/ja/docs/ARCHITECTURE.md b/docs/i18n/ja/docs/ARCHITECTURE.md index ad9129dc22..5a17df039a 100644 --- a/docs/i18n/ja/docs/ARCHITECTURE.md +++ b/docs/i18n/ja/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_最終更新日: 2026-03-28_## Executive Summary -OmniRoute は、Next.js 上に構築されたローカル AI ルーティング ゲートウェイおよびダッシュボードです。 -単一の OpenAI 互換エンドポイント (`/v1/*`) を提供し、変換、フォールバック、トークン更新、および使用状況追跡を使用して複数の上流プロバイダー間でトラフィックをルーティングします。 -コア機能: +_Last updated: 2026-03-28_ -- CLI/ツール用の OpenAI 互換 API サーフェス (28 プロバイダー) -- プロバイダ形式間でのリクエスト/レスポンスの変換 -- モデル コンボ フォールバック (マルチモデル シーケンス) -- アカウントレベルのフォールバック (プロバイダーごとにマルチアカウント) -- OAuth + APIキープロバイダ接続管理 -- `/v1/embeddings` による埋め込み生成 (6 プロバイダー、9 モデル) -- `/v1/images/generations` によるイメージ生成 (4 プロバイダー、9 モデル) -- 推論モデルの Think タグ解析 (`...`) -- 厳密な OpenAI SDK 互換性のための応答のサニタイズ -- プロバイダー間の互換性のための役割の正規化 (開発者→システム、システム→ユーザー) -- 構造化出力変換 (json_schema → Gemini responseSchema) -- プロバイダー、キー、エイリアス、コンボ、設定、価格設定のローカル永続性 -- 使用量/コストの追跡とリクエストのロギング -- マルチデバイス/状態同期のためのオプションのクラウド同期 -- API アクセス制御用の IP 許可リスト/ブロックリスト -- 予算管理を考える (パススルー/自動/カスタム/アダプティブ) -- グローバル システム プロンプト インジェクション -- セッション追跡とフィンガープリンティング -- プロバイダー固有のプロファイルによるアカウントごとの強化されたレート制限 -- プロバイダーの回復力を高めるサーキット ブレーカー パターン -- ミューテックスロックによるアンチサンダーリング保護 -- 署名ベースのリクエスト重複排除キャッシュ -- ドメイン層: モデルの可用性、コスト ルール、フォールバック ポリシー、ロックアウト ポリシー -- ドメイン状態の永続性 (フォールバック、バジェット、ロックアウト、サーキット ブレーカー用の SQLite ライトスルー キャッシュ) -- リクエストを一元的に評価するためのポリシー エンジン (ロックアウト → 予算 → フォールバック) -- p50/p95/p99 レイテンシ集約を使用したテレメトリのリクエスト -- エンドツーエンド トレース用の相関 ID (X-Request-Id) -- API キーごとのオプトアウトによるコンプライアンス監査ログ -- LLM品質保証のための評価フレームワーク -- リアルタイムのサーキット ブレーカー ステータスを備えた Resilience UI ダッシュボード -- モジュラー OAuth プロバイダー (`src/lib/oauth/providers/` にある 12 個の個別モジュール) +## Executive Summary -プライマリ ランタイム モデル: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- `src/app/api/*` の下の Next.js アプリ ルートは、ダッシュボード API と互換性 API の両方を実装します。 -- `src/sse/*` + `open-sse/*` の共有 SSE/ルーティング コアは、プロバイダーの実行、変換、ストリーミング、フォールバック、および使用を処理します。## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- ローカルゲートウェイランタイム -- ダッシュボード管理 API -- プロバイダー認証とトークンの更新 -- 翻訳と SSE ストリーミングのリクエスト -- ローカル状態 + 使用状況の永続性 -- オプションのクラウド同期オーケストレーション### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- `NEXT_PUBLIC_CLOUD_URL` の背後にあるクラウド サービスの実装 -- ローカル プロセス外のプロバイダー SLA/コントロール プレーン -- 外部 CLI バイナリ自体 (Claude CLI、Codex CLI など)## Dashboard Surface (Current) +### Out of Scope -`src/app/(dashboard)/dashboard/` の下のメイン ページ: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — クイック スタート + プロバイダーの概要 -- `/dashboard/endpoint` — エンドポイント プロキシ + MCP + A2A + API エンドポイント タブ -- `/dashboard/providers` — プロバイダー接続と認証情報 -- `/dashboard/combos` — コンボ戦略、テンプレート、モデルルーティングルール -- `/dashboard/costs` — コストの集計と価格の可視性 -- `/dashboard/analytics` — 使用状況の分析と評価 -- `/dashboard/limits` — クォータ/レート制御 -- `/dashboard/cli-tools` — CLI オンボーディング、ランタイム検出、構成生成 -- `/dashboard/agents` — 検出された ACP エージェント + カスタム エージェント登録 -- `/dashboard/media` — 画像/ビデオ/音楽のプレイグラウンド -- `/dashboard/search-tools` — 検索プロバイダーのテストと履歴 -- `/dashboard/health` — 稼働時間、サーキットブレーカー、レート制限 -- `/dashboard/logs` — リクエスト/プロキシ/監査/コンソール ログ -- `/dashboard/settings` — システム設定タブ (一般、ルーティング、コンボのデフォルトなど) -- `/dashboard/api-manager` — API キーのライフサイクルとモデルの権限## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -メインディレクトリ: +Main directories: -- 互換性 API 用の `src/app/api/v1/*` および `src/app/api/v1beta/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs - `src/app/api/*` for management/configuration APIs -- 次に、`next.config.mjs` を書き換えて `/v1/*` を `/api/v1/*` にマップします。 +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -重要な互換性ルート: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` — `custom: true` のカスタム モデルが含まれます -- `src/app/api/v1/embeddings/route.ts` — 埋め込み生成 (6 プロバイダー) -- `src/app/api/v1/images/generations/route.ts` — 画像生成 (Antigravity/Nebius を含む 4 つ以上のプロバイダー) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — プロバイダーごとの専用チャット -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — プロバイダーごとの専用埋め込み -- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — プロバイダーごとの専用イメージ +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` - `src/app/api/v1beta/models/[...path]/route.ts` -管理ドメイン: +Management domains: -- 認証/設定: `src/app/api/auth/*`、`src/app/api/settings/*` -- プロバイダー/接続: `src/app/api/providers*` -- プロバイダーノード: `src/app/api/provider-nodes*` -- カスタム モデル: `src/app/api/provider-models` (GET/POST/DELETE) -- モデルカタログ: `src/app/api/models/route.ts` (GET) -- プロキシ設定: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- キー/エイリアス/コンボ/価格: `src/app/api/keys*`、`src/app/api/models/alias`、`src/app/api/combos*`、`src/app/api/pricing` -- 使用法: `src/app/api/usage/*` -- 同期/クラウド: `src/app/api/sync/*`、`src/app/api/cloud/*` -- CLI ツールヘルパー: `src/app/api/cli-tools/*` -- IP フィルター: `src/app/api/settings/ip-filter` (GET/PUT) -- 思考予算: `src/app/api/settings/ Thinking-budget` (GET/PUT) -- システムプロンプト: `src/app/api/settings/system-prompt` (GET/PUT) -- セッション: `src/app/api/sessions` (GET) -- レート制限: `src/app/api/rate-limits` (GET) -- 復元力: `src/app/api/resilience` (GET/PATCH) — プロバイダー プロファイル、サーキット ブレーカー、レート制限の状態 -- レジリエンスのリセット: `src/app/api/resilience/reset` (POST) — ブレーカー + クールダウンのリセット -- キャッシュ統計: `src/app/api/cache/stats` (GET/DELETE) -- モデルの可用性: `src/app/api/models/availability` (GET/POST) -- テレメトリ: `src/app/api/telemetry/summary` (GET) -- 予算: `src/app/api/usage/budget` (GET/POST) -- フォールバック チェーン: `src/app/api/fallback/chains` (GET/POST/DELETE) -- コンプライアンス監査: `src/app/api/compliance/audit-log` (GET) -- Evals: `src/app/api/evals` (GET/POST)、`src/app/api/evals/[suiteId]` (GET) -- ポリシー: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -メインフローモジュール: +## 2) SSE + Translation Core -- エントリ: `src/sse/handlers/chat.ts` -- コア オーケストレーション: `open-sse/handlers/chatCore.ts` -- プロバイダー実行アダプター: `open-sse/executors/*` -- フォーマット検出/プロバイダー設定: `open-sse/services/provider.ts` -- モデルの解析/解決: `src/sse/services/model.ts`、`open-sse/services/model.ts` -- アカウントフォールバックロジック: `open-sse/services/accountFallback.ts` -- 翻訳レジストリ: `open-sse/translator/index.ts` -- ストリーム変換: `open-sse/utils/stream.ts`、`open-sse/utils/streamHandler.ts` -- 使用状況の抽出/正規化: `open-sse/utils/usageTracking.ts` -- Think タグ パーサー: `open-sse/utils/thinkTagParser.ts` -- 埋め込みハンドラー: `open-sse/handlers/embeddings.ts` -- 埋め込みプロバイダー レジストリ: `open-sse/config/embeddingRegistry.ts` -- 画像生成ハンドラー: `open-sse/handlers/imageGeneration.ts` -- イメージ プロバイダー レジストリ: `open-sse/config/imageRegistry.ts` -- レスポンスのサニタイズ: `open-sse/handlers/responseSanitizer.ts` -- ロールの正規化: `open-sse/services/roleNormalizer.ts` +Main flow modules: -サービス (ビジネス ロジック): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- アカウントの選択/スコアリング: `open-sse/services/accountSelector.ts` -- コンテキストのライフサイクル管理: `open-sse/services/contextManager.ts` -- IP フィルターの適用: `open-sse/services/ipFilter.ts` -- セッション追跡: `open-sse/services/sessionManager.ts` -- 重複排除のリクエスト: `open-sse/services/signatureCache.ts` -- システム プロンプト インジェクション: `open-sse/services/systemPrompt.ts` -- 思考予算管理: `open-sse/services/ ThinkingBudget.ts` -- ワイルドカード モデル ルーティング: `open-sse/services/wildcardRouter.ts` -- レート制限管理: `open-sse/services/rateLimitManager.ts` -- サーキットブレーカー: `open-sse/services/circuitBreaker.ts` +Services (business logic): -ドメイン層モジュール: +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- モデルの可用性: `src/lib/domain/modelAvailability.ts` -- コストルール/予算: `src/lib/domain/costRules.ts` -- フォールバック ポリシー: `src/lib/domain/fallbackPolicy.ts` -- コンボリゾルバー: `src/lib/domain/comboResolver.ts` -- ロックアウト ポリシー: `src/lib/domain/lockoutPolicy.ts` -- ポリシー エンジン: `src/domain/policyEngine.ts` — 集中ロックアウト → 予算 → フォールバック評価 -- エラーコードカタログ: `src/lib/domain/errorCodes.ts` -- リクエストID: `src/lib/domain/requestId.ts` -- フェッチタイムアウト: `src/lib/domain/fetchTimeout.ts` -- テレメトリのリクエスト: `src/lib/domain/requestTelemetry.ts` -- コンプライアンス/監査: `src/lib/domain/compliance/index.ts` -- 評価ランナー: `src/lib/domain/evalRunner.ts` -- ドメイン状態の永続性: `src/lib/db/domainState.ts` — フォールバック チェーン、予算、コスト履歴、ロックアウト状態、サーキット ブレーカー用の SQLite CRUD +Domain layer modules: -OAuth プロバイダー モジュール (`src/lib/oauth/providers/` にある 12 個の個別のファイル): +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -- レジストリ インデックス: `src/lib/oauth/providers/index.ts` -- 個々のプロバイダー: `claude.ts`、`codex.ts`、`gemini.ts`、`antigravity.ts`、`qoder.ts`、`qwen.ts`、`kimi-coding.ts`、`github.ts`、`kiro.ts`、`cursor.ts`、`kilocode.ts`、 「クライン.ts」 -- 薄いラッパー: `src/lib/oauth/providers.ts` — 個々のモジュールからの再エクスポート## 3) Persistence Layer +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -プライマリ状態 DB (SQLite): +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -- コアインフラ: `src/lib/db/core.ts` (better-sqlite3、移行、WAL) -- ファサードの再エクスポート: `src/lib/localDb.ts` (呼び出し元用の薄い互換性レイヤー) -- ファイル: `${DATA_DIR}/storage.sqlite` (設定されている場合は `$XDG_CONFIG_HOME/omniroute/storage.sqlite`、それ以外の場合は `~/.omniroute/storage.sqlite`) -- エンティティ (テーブル + KV 名前空間): ProviderConnections、providerNodes、modelAliases、コンボ、apiKeys、設定、価格設定、**customModels**、**proxyConfig**、**ipFilter**、**ThinkingBudget**、**systemPrompt** +## 3) Persistence Layer -使用の永続性: +Primary state DB (SQLite): -- ファサード: `src/lib/usageDb.ts` (`src/lib/usage/*` にある分解されたモジュール) -- `storage.sqlite` 内の SQLite テーブル: `usage_history`、`call_logs`、`proxy_logs` -- オプションのファイル アーティファクトは互換性/デバッグのために残ります (`${DATA_DIR}/log.txt`、`${DATA_DIR}/call_logs/`、`/logs/...`) -- 従来の JSON ファイルが存在する場合、起動時の移行によって SQLite に移行されます +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -ドメイン状態 DB (SQLite): +Usage persistence: -- `src/lib/db/domainState.ts` — ドメイン状態の CRUD 操作 -- テーブル (`src/lib/db/core.ts` で作成): `domain_fallback_chains`、`domain_budgets`、`domain_cost_history`、`domain_lockout_state`、`domain_circuit_breakers` -- ライトスルー キャッシュ パターン: メモリ内マップは実行時に権限を持ちます。変更は SQLite に同期的に書き込まれます。状態はコールド スタート時に DB から復元されます## 4) Auth + Security Surfaces +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- ダッシュボード Cookie 認証: `src/proxy.ts`、`src/app/api/auth/login/route.ts` -- APIキーの生成/検証: `src/shared/utils/apiKey.ts` -- プロバイダーのシークレットは「providerConnections」エントリに保持されます -- `open-sse/utils/proxyFetch.ts` (環境変数) および `open-sse/utils/networkProxy.ts` (プロバイダーごとまたはグローバルに構成可能) による送信プロキシのサポート## 5) Cloud Sync +Domain State DB (SQLite): -- スケジューラの初期化: `src/lib/initCloudSync.ts`、`src/shared/services/initializeCloudSync.ts`、`src/shared/services/modelSyncScheduler.ts` -- 定期タスク: `src/shared/services/cloudSyncScheduler.ts` -- 定期タスク: `src/shared/services/modelSyncScheduler.ts` -- 制御ルート: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -フォールバックの決定は、ステータス コードとエラー メッセージのヒューリスティックを使用して「open-sse/services/accountFallback.ts」によって行われます。コンボ ルーティングにより、追加のガードが 1 つ追加されます。アップストリームのコンテンツ ブロックやロール検証の失敗など、プロバイダー スコープの 400 はモデル ローカルの失敗として扱われるため、後のコンボ ターゲットは引き続き実行できます。## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -ライブ トラフィック中の更新は、実行プログラム `refreshCredentials()` を介して `open-sse/handlers/chatCore.ts` 内で実行されます。## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -クラウドが有効になっている場合、定期的な同期は「CloudSyncScheduler」によってトリガーされます。## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -物理ストレージ ファイル: +Physical storage files: -- プライマリ ランタイム DB: `${DATA_DIR}/storage.sqlite` -- リクエストログ行: `${DATA_DIR}/log.txt` (互換/デバッグアーティファクト) -- 構造化された通話ペイロード アーカイブ: `${DATA_DIR}/call_logs/` -- オプションのトランスレーター/リクエスト デバッグ セッション: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`、`src/app/api/v1beta/*`: 互換性 API -- `src/app/api/v1/providers/[provider]/*`: プロバイダーごとの専用ルート (チャット、埋め込み、画像) -- `src/app/api/providers*`: プロバイダー CRUD、検証、テスト -- `src/app/api/provider-nodes*`: カスタム互換ノード管理 -- `src/app/api/provider-models`: カスタム モデル管理 (CRUD) -- `src/app/api/models/route.ts`: モデル カタログ API (エイリアス + カスタム モデル) -- `src/app/api/oauth/*`: OAuth/device-code フロー -- `src/app/api/keys*`: ローカル API キーのライフサイクル -- `src/app/api/models/alias`: エイリアス管理 -- `src/app/api/combos*`: フォールバックコンボ管理 -- `src/app/api/pricing`: コスト計算のための価格設定の上書き -- `src/app/api/settings/proxy`: プロキシ設定 (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: 送信プロキシ接続テスト (POST) -- `src/app/api/usage/*`: 使用状況とログ API -- `src/app/api/sync/*` + `src/app/api/cloud/*`: クラウド同期およびクラウド対応ヘルパー -- `src/app/api/cli-tools/*`: ローカル CLI 設定ライター/チェッカー -- `src/app/api/settings/ip-filter`: IP 許可リスト/ブロックリスト (GET/PUT) -- `src/app/api/settings/ Thinking-budget`: 思考トークンの予算設定 (GET/PUT) -- `src/app/api/settings/system-prompt`: グローバル システム プロンプト (GET/PUT) -- `src/app/api/sessions`: アクティブなセッションのリスト (GET) -- `src/app/api/rate-limits`: アカウントごとのレート制限ステータス (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: リクエスト解析、コンボ処理、アカウント選択ループ -- `open-sse/handlers/chatCore.ts`: 変換、実行プログラムのディスパッチ、再試行/リフレッシュ処理、ストリームのセットアップ -- `open-sse/executors/*`: プロバイダー固有のネットワークと形式の動作### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: トランスレーターのレジストリとオーケストレーション -- トランスレータのリクエスト: `open-sse/translator/request/*` -- 応答トランスレーター: `open-sse/translator/response/*` -- フォーマット定数: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: SQLite での永続的な設定/状態とドメインの永続化 -- `src/lib/localDb.ts`: DB モジュールの互換性再エクスポート -- `src/lib/usageDb.ts`: SQLite テーブルの上部にある使用履歴/通話ログのファサード## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -各プロバイダーには、`BaseExecutor` (`open-sse/executors/base.ts` 内) を拡張する特殊なエグゼキューターがあり、URL の構築、ヘッダーの構築、指数バックオフによる再試行、資格情報の更新フック、および `execute()` オーケストレーション メソッドを提供します。 +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| 執行者 | プロバイダー | 特殊な取り扱い | -| -------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ---------------------------------------------------------------------------- | -| `デフォルトエグゼキュータ` | OpenAI、Claude、Gemini、Qwen、Qoder、OpenRouter、GLM、Kimi、MiniMax、DeepSeek、Groq、xAI、Mistral、Perplexity、Togetter、Fireworks、Cerebros、Cohere、NVIDIA | プロバイダーごとの動的 URL/ヘッダー構成 | -| `反重力エグゼキューター` | Google 反重力 | カスタム プロジェクト/セッション ID、解析後の再試行 | -| `CodexExecutor` | OpenAI コーデックス | システム命令を挿入し、推論努力を強制する | -| `CursorExecutor` | カーソルIDE | ConnectRPC プロトコル、Protobuf エンコーディング、チェックサムによる要求署名 | -| `GithubExecutor` | GitHub コパイロット | コパイロット トークンの更新、VSCode を模倣したヘッダー | -| `キロエグゼキューター` | AWS CodeWhisperer/Kiro | AWS EventStream バイナリ形式 → SSE 変換 | -| `GeminiCLIExecutor` | ジェミニ CLI | Google OAuth トークンの更新サイクル | +### Persistence -他のすべてのプロバイダー (カスタム互換ノードを含む) は `DefaultExecutor` を使用します。## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| プロバイダー | フォーマット | 認証 | ストリーム | 非ストリーム | トークンのリフレッシュ | 使用法 API | -| --------------------- | ------------------ | ----------------------------- | ----------------------- | ------------ | ---------------------- | ----------------------------- | ------------------------------ | -| クロード | クロード | APIキー/OAuth | ✅ | ✅ | ✅ | ⚠️管理者のみ | -| ジェミニ | ジェミニ | APIキー/OAuth | ✅ | ✅ | ✅ | ⚠️クラウドコンソール | -| ジェミニ CLI | ジェミニクリ | OAuth | ✅ | ✅ | ✅ | ⚠️クラウドコンソール | -| 反重力 | 反重力 | OAuth | ✅ | ✅ | ✅ | ✅ フルクォータ API | -| オープンAI | オープンナイ | APIキー | ✅ | ✅ | ❌ | ❌ | -| コーデックス | オープンナイの応答 | OAuth | ✅強制 | ❌ | ✅ | ✅ レート制限 | -| GitHub コパイロット | オープンナイ | OAuth + コパイロット トークン | ✅ | ✅ | ✅ | ✅ クォータのスナップショット | -| カーソル | カーソル | カスタムチェックサム | ✅ | ✅ | ❌ | ❌ | -| キロ | キロ | AWS SSO OIDC | ✅ (イベントストリーム) | ❌ | ✅ | ✅ 使用制限 | -| クウェン | オープンナイ | OAuth | ✅ | ✅ | ✅ | ⚠️リクエストに応じて | -| コーダー | オープンナイ | OAuth (基本) | ✅ | ✅ | ✅ | ⚠️リクエストに応じて | -| オープンルーター | オープンナイ | APIキー | ✅ | ✅ | ❌ | ❌ | -| GLM/キミ/ミニマックス | クロード | APIキー | ✅ | ✅ | ❌ | ❌ | -| ディープシーク | オープンナイ | APIキー | ✅ | ✅ | ❌ | ❌ | -| グロク | オープンナイ | APIキー | ✅ | ✅ | ❌ | ❌ | -| xAI (グロック) | オープンナイ | APIキー | ✅ | ✅ | ❌ | ❌ | -| ミストラル | オープンナイ | APIキー | ✅ | ✅ | ❌ | ❌ | -| 困惑 | オープンナイ | APIキー | ✅ | ✅ | ❌ | ❌ | -| 一緒にAI | オープンナイ | APIキー | ✅ | ✅ | ❌ | ❌ | -| 花火AI | オープンナイ | APIキー | ✅ | ✅ | ❌ | ❌ | -| 大脳 | オープンナイ | APIキー | ✅ | ✅ | ❌ | ❌ | -| コヒア | オープンナイ | APIキー | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | オープンナイ | APIキー | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -検出されたソース形式は次のとおりです。 +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. -- 「オープンナイ」 -- 「openai-responses」 -- 「クロード」 -- 「ジェミニ」 +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | -対象となる形式は次のとおりです。 +All other providers (including custom compatible nodes) use the `DefaultExecutor`. -- OpenAI チャット/応答 -- クロード -- ジェミニ/ジェミニ-CLI/反重力エンベロープ -- キロ -- カーソル +## Provider Compatibility Matrix -翻訳では**OpenAI をハブ形式**として使用します。すべての変換は中間として OpenAI を経由します。``` +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: + +- `openai` +- `openai-responses` +- `claude` +- `gemini` + +Target formats include: + +- OpenAI chat/Responses +- Claude +- Gemini/Gemini-CLI/Antigravity envelope +- Kiro +- Cursor + +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -翻訳は、ソース ペイロードの形状とプロバイダーのターゲット形式に基づいて動的に選択されます。 +Additional processing layers in the translation pipeline: -翻訳パイプラインの追加の処理レイヤー: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**レスポンスのサニタイズ**— OpenAI 形式のレスポンス (ストリーミングと非ストリーミングの両方) から非標準フィールドを削除し、厳密な SDK コンプライアンスを確保します。 --**ロールの正規化**— 非 OpenAI ターゲットの「開発者」→「システム」を変換します。システムの役割を拒否するモデル (GLM、ERNIE) の `system` → `user` をマージします。 --**Think タグの抽出**— コンテンツから `...` ブロックを解析して `reasoning_content` フィールドに入れます --**構造化出力**— OpenAI の `response_format.json_schema` を Gemini の `responseMimeType` + `responseSchema` に変換します## Supported API Endpoints +## Supported API Endpoints -|エンドポイント |フォーマット |ハンドラー | +| Endpoint | Format | Handler | | -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | -| `POST /v1/chat/completions` | OpenAIチャット | `src/sse/handlers/chat.ts` | -| `POST /v1/messages` |クロードのメッセージ |同じハンドラー (自動検出) | -| `POST /v1/responses` | OpenAI の応答 | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/embeddings` | OpenAI 埋め込み | `open-sse/handlers/embeddings.ts` | -| `GET /v1/embeddings` |モデル一覧 | APIルート | -| `POST /v1/images/世代` | OpenAI 画像 | `open-sse/handlers/imageGeneration.ts` | -| `GET /v1/images/世代` |モデル一覧 | APIルート | -| `POST /v1/providers/{provider}/chat/completions` | OpenAIチャット |モデル検証を備えた専用のプロバイダーごと | -| `POST /v1/providers/{provider}/embeddings` | OpenAI 埋め込み |モデル検証を備えた専用のプロバイダーごと | -| `POST /v1/providers/{provider}/images/世代` | OpenAI 画像 |モデル検証を備えた専用のプロバイダーごと | -| `POST /v1/messages/count_tokens` |クロードトークン数 | APIルート | -| `GET /v1/models` | OpenAI Models list | API ルート (チャット + 埋め込み + 画像 + カスタム モデル) | -| `/api/models/catalog` を取得する |カタログ |プロバイダー + タイプごとにグループ化されたすべてのモデル | -| `POST /v1beta/models/*:streamGenerateContent` |双子座出身 | APIルート | -| `GET/PUT/DELETE /api/settings/proxy` |プロキシ構成 |ネットワークプロキシ構成 | -| `POST /api/settings/proxy/test` |プロキシ接続 |プロキシの正常性/接続テスト エンドポイント | -| `GET/POST/DELETE /api/provider-models` |プロバイダーモデル |カスタムおよび管理された利用可能なモデルを裏付けるプロバイダー モデルのメタデータ |## Bypass Handler +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -バイパス ハンドラー (`open-sse/utils/bypassHandler.ts`) は、Claude CLI からの既知の「使い捨て」リクエスト (ウォームアップ ping、タイトル抽出、トークン カウント) をインターセプトし、アップストリーム プロバイダー トークンを消費せずに**偽の応答**を返します。これは、`User-Agent` に `claude-cli` が含まれている場合にのみトリガーされます。## Request Logger Pipeline +## Bypass Handler -リクエスト ロガー (`open-sse/utils/requestLogger.ts`) は 7 段階のデバッグ ロギング パイプラインを提供します。デフォルトでは無効になっており、`ENABLE_REQUEST_LOGS=true` で有効になります。``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -ファイルはリクエスト セッションごとに `/logs//` に書き込まれます。## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- 一時的/レート/認証エラー時のプロバイダー アカウントのクールダウン -- リクエストが失敗する前のアカウントのフォールバック -- 現在のモデル/プロバイダー パスが枯渇した場合のコンボ モデル フォールバック## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- 更新可能なプロバイダーの事前チェックと再試行による更新 -- コア パスでの更新試行後の 401/403 再試行## 3) Stream Safety +## 2) Token Expiry -- 切断対応ストリーム コントローラー -- ストリーム終了フラッシュと「[DONE]」処理を備えた変換ストリーム -- プロバイダーの使用量メタデータが欠落している場合の使用量推定フォールバック## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- 同期エラーが表面化しましたが、ローカル ランタイムは継続します -- スケジューラには再試行可能なロジックがありますが、定期的な実行では現在、デフォルトで単一試行同期が呼び出されます。## 5) Data Integrity +## 3) Stream Safety -- SQLite スキーマの移行と起動時の自動アップグレード フック -- レガシー JSON → SQLite 移行互換パス## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -実行時の可視性ソース: +## 4) Cloud Sync Degradation -- `src/sse/utils/logger.ts` からのコンソール ログ -- SQLite でのリクエストごとの使用状況の集計 (`usage_history`、`call_logs`、`proxy_logs`) -- `settings.detailed_logs_enabled=true` の場合、SQLite での 4 段階の詳細なペイロード キャプチャ (`request_detail_logs`) -- `log.txt` 内のテキスト形式のリクエスト ステータス ログ (オプション/互換性) -- `ENABLE_REQUEST_LOGS=true` の場合、`logs/` の下にあるオプションの詳細なリクエスト/変換ログ -- UI 消費のためのダッシュボード使用エンドポイント (`/api/usage/*`) +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -詳細なリクエスト ペイロード キャプチャでは、ルーティングされた呼び出しごとに最大 4 つの JSON ペイロード ステージが保存されます。 +## 5) Data Integrity -- クライアントから受信した生のリクエスト -- 翻訳されたリクエストは実際に上流に送信されます -- プロバイダーの応答は JSON として再構築されます。ストリーミングされた応答は、最終的なサマリーとストリーム メタデータに圧縮されます。 -- OmniRoute によって返される最終クライアント応答。ストリーミングされた応答は同じコンパクトな概要形式に保存されます## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- JWT シークレット (`JWT_SECRET`) はダッシュボード セッションの Cookie 検証/署名を保護します -- 初回実行プロビジョニング用に初期パスワード ブートストラップ (`INITIAL_PASSWORD`) を明示的に構成する必要があります -- API キー HMAC シークレット (`API_KEY_SECRET`) は、生成されたローカル API キー形式を保護します -- プロバイダーのシークレット (API キー/トークン) はローカル DB に保存され、ファイルシステム レベルで保護される必要があります。 -- クラウド同期エンドポイントは、API キー認証 + マシン ID セマンティクスに依存します。## Environment and Runtime Matrix +## Observability and Operational Signals -コードによってアクティブに使用される環境変数: +Runtime visibility sources: -- アプリ/認証: `JWT_SECRET`、`INITIAL_PASSWORD` -- ストレージ: `DATA_DIR` -- 互換性のあるノードの動作: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- オプションのストレージ ベース オーバーライド (Linux/macOS `DATA_DIR` が設定されていない場合): `XDG_CONFIG_HOME` -- セキュリティハッシュ: `API_KEY_SECRET`、`MACHINE_ID_SALT` -- ロギング: `ENABLE_REQUEST_LOGS` -- 同期/クラウド URL 指定: `NEXT_PUBLIC_BASE_URL`、`NEXT_PUBLIC_CLOUD_URL` -- 送信プロキシ: `HTTP_PROXY`、`HTTPS_PROXY`、`ALL_PROXY`、`NO_PROXY` および小文字のバリアント -- SOCKS5 機能フラグ: `ENABLE_SOCKS5_PROXY`、`NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- プラットフォーム/ランタイム ヘルパー (アプリ固有の構成ではない): `APPDATA`、`NODE_ENV`、`PORT`、`HOSTNAME`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` と `localDb` は、レガシー ファイル移行と同じベース ディレクトリ ポリシー (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) を共有します。 -2. `/api/v1/route.ts` は、セマンティック ドリフトを避けるために、`/api/v1/models` (`src/app/api/v1/models/catalog.ts`) によって使用されるのと同じ統合カタログ ビルダーに委任します。 -3. リクエスト ロガーは有効な場合、完全なヘッダー/本文を書き込みます。ログ ディレクトリを機密として扱います。 -4. クラウドの動作は、正しい「NEXT_PUBLIC_BASE_URL」とクラウド エンドポイントの到達可能性に依存します。 -5. The `open-sse/` directory is published as the `@omniroute/open-sse`**npm workspace package**.ソース コードは `@omniroute/open-sse/...` 経由でインポートします (Next.js `transpilePackages` によって解決されます)。このドキュメントのファイル パスでは、一貫性を保つために引き続きディレクトリ名 `open-sse/` が使用されます。 -6. ダッシュボードのグラフでは、**Recharts**(SVG ベース) を使用して、アクセスしやすく対話型の分析を視覚化します (モデル使用状況の棒グラフ、成功率を示すプロバイダーの内訳表)。 -7. E2E テストは**Playwright**(`tests/e2e/`) を使用し、`npm run test:e2e` 経由で実行します。単体テストは**Node.js テスト ランナー**(`tests/unit/`) を使用し、`npm run test:unit` 経由で実行します。 `src/` の下のソース コードは**TypeScript**(`.ts`/`.tsx`) です。 「open-sse/」ワークスペースは JavaScript (「.js」) のままです。 -8. 設定ページは 5 つのタブで構成されています: セキュリティ、ルーティング (6 つのグローバル戦略: フィルファースト、ラウンドロビン、p2c、ランダム、最小使用、コスト最適化)、復元力 (編集可能なレート制限、サーキット ブレーカー、ポリシー)、AI (思考予算、システム プロンプト、プロンプト キャッシュ)、詳細 (プロキシ)。## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- ソースからビルド: `npm run build` -- Docker イメージをビルドします: `docker build -tomniroute .` -- サービスを開始して以下を確認します。 -- `/api/settings` を取得します -- 「/api/v1/models を取得」 -- CLI ターゲットのベース URL は、「PORT=20128」の場合は「http://:20128/v1」である必要があります。 +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` +- `GET /api/v1/models` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/ja/docs/FEATURES.md b/docs/i18n/ja/docs/FEATURES.md index 06086471d4..b3faf05cb9 100644 --- a/docs/i18n/ja/docs/FEATURES.md +++ b/docs/i18n/ja/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -OmniRoute ダッシュボードの各セクションへの視覚的なガイド。--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -AI プロバイダー接続の管理: OAuth プロバイダー (Claude Code、Codex、Gemini CLI)、API キー プロバイダー (Groq、DeepSeek、OpenRouter)、および無料プロバイダー (Qoder、Qwen、Kiro)。 Kiro アカウントには、クレジット残高の追跡機能が含まれています。残存クレジット、合計許容量、更新日は、「ダッシュボード」→「使用状況」で確認できます。![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -6 つの戦略 (優先順位、重み付け、ラウンドロビン、ランダム、最小使用、コスト最適化) を使用してモデル ルーティング コンボを作成します。各コンボは自動フォールバックを使用して複数のモデルをチェーンし、クイック テンプレートと準備状況チェックが含まれます。![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -トークン消費量、コスト見積もり、アクティビティヒートマップ、週次分布グラフ、プロバイダーごとの内訳を含む包括的な使用状況分析。![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -リアルタイム監視: 稼働時間、メモリ、バージョン、遅延パーセンタイル (p50/p95/p99)、キャッシュ統計、プロバイダーのサーキット ブレーカーの状態。![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -API 変換をデバッグするための 4 つのモード:**プレイグラウンド**(フォーマット コンバーター)、**チャット テスター**(ライブ リクエスト)、**テスト ベンチ**(バッチ テスト)、**ライブ モニター**(リアルタイム ストリーム)。![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -ダッシュボードから直接任意のモデルをテストします。プロバイダー、モデル、エンドポイントを選択し、Monaco Editor でプロンプトを作成し、リアルタイムで応答をストリーミングし、ストリームの途中で中止し、タイミング メトリクスを表示します。--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -ダッシュボード全体のカスタマイズ可能なカラーテーマ。 7 つのプリセット色 (コーラル、ブルー、レッド、グリーン、バイオレット、オレンジ、シアン) から選択するか、任意の 16 進カラーを選択してカスタム テーマを作成します。ライト、ダーク、システム モードをサポートします。--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -タブのある包括的な設定パネル: +Comprehensive settings panel with tabs: --**一般**— システム ストレージ、バックアップ管理 (データベースのエクスポート/インポート) -**外観**— テーマセレクター (ダーク/ライト/システム)、カラーテーマのプリセットとカスタムカラー、ヘルスログの表示、サイドバー項目の表示コントロール -**セキュリティ**- API エンドポイント保護、カスタム プロバイダー ブロック、IP フィルタリング、セッション情報 -**ルーティング**— モデルのエイリアス、バックグラウンド タスクの劣化 -**回復力**— レート制限の永続化、サーキット ブレーカーの調整、禁止アカウントの自動無効化、プロバイダーの有効期限の監視 -**上級**- 構成の上書き、構成監査証跡、フォールバック劣化モード![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -AI コーディング ツールのワンクリック構成: Claude Code、Codex CLI、Gemini CLI、OpenClaw、Kilo Code、Antigravity、Cline、Continue、Cursor、Factory Droid。自動構成の適用/リセット、接続プロファイル、モデル マッピングを備えています。![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -CLI エージェントを検出および管理するためのダッシュボード。 14 の組み込みエージェント (Codex、Claude、Goose、Gemini CLI、OpenClaw、Aider、OpenCode、Cline、Qwen Code、ForgeCode、Amazon Q、Open Interpreter、Cursor CLI、Warp) のグリッドを表示します。 +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**インストール ステータス**- インストール済み / バージョン検出で見つからない -**プロトコル バッジ**— stdio、HTTP など。-**カスタム エージェント**- フォーム経由で CLI ツールを登録します (名前、バイナリ、バージョン コマンド、生成引数)。-**CLI フィンガープリント マッチング**— プロバイダーごとに切り替えてネイティブ CLI リクエストの署名を照合し、プロキシ IP を維持しながら禁止リスクを軽減します--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -ダッシュボードから画像、ビデオ、音楽を生成します。 OpenAI、xAI、Togetter、Hyperbolic、SD WebUI、ComfyUI、AnimateDiff、Stable Audio Open、および MusicGen をサポートします。--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -プロバイダー、モデル、アカウント、API キーによるフィルタリングを備えたリアルタイムのリクエストログ。ステータス コード、トークンの使用状況、待ち時間、応答の詳細を表示します。![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -機能の内訳を含む統合 API エンドポイント: チャット完了、応答 API、埋め込み、画像生成、再ランキング、音声文字起こし、テキスト読み上げ、モデレーション、登録された API キー。 Cloudflare Quick Tunnelの統合とリモートアクセスのためのクラウドプロキシのサポート。![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -API キーを作成、スコープ設定、取り消します。各キーは、フルアクセスまたは読み取り専用のアクセス許可を持つ特定のモデル/プロバイダーに制限できます。使用状況追跡による視覚的なキー管理。--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -アクション タイプ、アクター、ターゲット、IP アドレス、およびタイムスタンプによるフィルタリングによる管理アクションの追跡。完全なセキュリティ イベント履歴。--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Windows、macOS、Linux 用のネイティブ Electron デスクトップ アプリ。システム トレイの統合、オフライン サポート、自動更新、ワンクリック インストールを備えたスタンドアロン アプリケーションとして OmniRoute を実行します。 +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -主な特徴: +Key features: -- サーバー準備状況ポーリング (コールド スタート時に空白画面なし) -- ポート管理を備えたシステムトレイ -- コンテンツセキュリティポリシー -- 単一インスタンスのロック -- 再起動時の自動更新 -- プラットフォーム条件付き UI (macOS 信号機、Windows/Linux のデフォルトのタイトルバー) -- 強化された Electron ビルド パッケージ化 - スタンドアロン バンドル内のシンボリックリンクされた `node_modules` がパッケージ化前に検出されて拒否され、ビルド マシン (v2.5.5 以降) へのランタイムの依存関係が防止されます。 +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 完全なドキュメントについては、[`electron/README.md`](../electron/README.md) を参照してください。 +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/ja/docs/TROUBLESHOOTING.md b/docs/i18n/ja/docs/TROUBLESHOOTING.md index 17e37220a3..54160df848 100644 --- a/docs/i18n/ja/docs/TROUBLESHOOTING.md +++ b/docs/i18n/ja/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -OmniRoute の一般的な問題と解決策。--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| 問題 | ソリューション | -| ----------------------------------------- | -------------------------------------------------------------------------------------------- | --- | -| 最初のログインが機能しない | `.env` に `INITIAL_PASSWORD` を設定します (ハードコーディングされたデフォルトはありません)。 | -| ダッシュボードが間違ったポートで開きます | `PORT=20128` と `NEXT_PUBLIC_BASE_URL=http://localhost:20128` を設定します。 | -| `logs/` の下にリクエスト ログがありません | `ENABLE_REQUEST_LOGS=true` を設定します。 | -| EACCES: 許可が拒否されました | `DATA_DIR=/path/to/writable/dir` を設定して `~/.omniroute` をオーバーライドします。 | -| ルーティング戦略が保存されない | v1.4.11+ に更新 (設定永続性のための Zod スキーマ修正) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**原因:**プロバイダーの割り当てが枯渇しました。 +**Cause:** Provider quota exhausted. -**修正:** +**Fix:** -1. ダッシュボードのクォータ トラッカーを確認する -2. フォールバック層とのコンボを使用する -3. より安価な/無料枠に切り替える### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**原因:**サブスクリプション割り当てを使い果たしました。 +### Rate Limiting -**修正:** +**Cause:** Subscription quota exhausted. -- フォールバックを追加: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2- Thinking` -- 安価なバックアップとして GLM/MiniMax を使用する### OAuth Token Expired +**Fix:** -OmniRoute はトークンを自動更新します。問題が解決しない場合: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. ダッシュボード → プロバイダー → 再接続 -2. プロバイダー接続を削除して再度追加します。--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. 「BASE_URL」が実行中のインスタンスを指していることを確認します (例: 「http://localhost:20128」) -2. `CLOUD_URL` がクラウド エンドポイント (例: `https://omniroute.dev`) を指していることを確認します。 -3. `NEXT_PUBLIC_*` 値をサーバー側の値と一致させます。### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**症状:**非ストリーミング呼び出しのクラウド エンドポイントで「予期しないトークン 'd'...」が発生します。 +### Cloud `stream=false` Returns 500 -**原因:**クライアントが JSON を期待しているのに、アップストリームは SSE ペイロードを返します。 +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**回避策:**クラウドの直接呼び出しには「stream=true」を使用します。ローカル ランタイムには SSE→JSON フォールバックが含まれます。### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. ローカル ダッシュボード (`/api/keys`) から新しいキーを作成します。 -2. クラウド同期を実行します: [クラウドを有効にする] → [今すぐ同期] -3. 古い/非同期キーはクラウド上でも「401」を返す可能性がある--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. 実行時フィールドを確認します。 `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. ポータブル モードの場合: イメージ ターゲット `runner-cli` (バンドルされた CLI) を使用します。 -3. ホスト マウント モードの場合: 「CLI_EXTRA_PATHS」を設定し、ホストの bin ディレクトリを読み取り専用としてマウントします。 -4. `installed=true` および `runnable=false` の場合: バイナリは見つかりましたが、ヘルスチェックに失敗しました### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,13 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1.「ダッシュボード」→「使用状況」で使用状況統計を確認します。2. Switch primary model to GLM/MiniMax 3. 重要でないタスクには無料枠 (Gemini CLI、Qoder) を使用する 4. API キーごとにコスト予算を設定します: [ダッシュボード] → [API キー] → [予算]--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -`.env` ファイルに `ENABLE_REQUEST_LOGS=true` を設定します。ログは「logs/」ディレクトリの下に表示されます。### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -97,93 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- メイン状態: `${DATA_DIR}/storage.sqlite` (プロバイダー、コンボ、エイリアス、キー、設定) -- 使用法: `storage.sqlite` 内の SQLite テーブル (`usage_history`、`call_logs`、`proxy_logs`) + オプションの `${DATA_DIR}/log.txt` および `${DATA_DIR}/call_logs/` -- リクエストログ: `/logs/...` (`ENABLE_REQUEST_LOGS=true`の場合)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -プロバイダーのサーキット ブレーカーが OPEN の場合、リクエストはクールダウンが期限切れになるまでブロックされます。 +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**修正:** +**Fix:** -1.**ダッシュボード → 設定 → レジリエンス**に移動します 2. 影響を受けるプロバイダーのサーキット ブレーカー カードを確認します。3. [**すべてリセット**] をクリックしてすべてのブレーカーをクリアするか、クールダウンが期限切れになるまで待ちます。4. リセットする前に、プロバイダーが実際に利用可能であることを確認します。### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -プロバイダーが繰り返し OPEN 状態になる場合: +### Provider keeps tripping the circuit breaker -1.**ダッシュボード → ヘルス → プロバイダーのヘルス**で障害パターンを確認します。2.**[設定] → [復元力] → [プロバイダー プロファイル]**に移動し、失敗のしきい値を増やします。3. プロバイダーが API 制限を変更したか、再認証が必要かどうかを確認します。4. レイテンシーテレメトリを確認します - レイテンシーが長いとタイムアウトベースのエラーが発生する可能性があります--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- 正しいプレフィックスを使用していることを確認してください: `deepgram/nova-3` または `assemblyai/best` -**「ダッシュボード」→「プロバイダー」**でプロバイダーが接続されていることを確認します。### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- サポートされているオーディオ形式を確認します: `mp3`、`wav`、`m4a`、`flac`、`ogg`、`webm` -- ファイル サイズがプロバイダーの制限内であることを確認します (通常は < 25MB) -- プロバイダー カードのプロバイダー API キーの有効性を確認します。--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -**ダッシュボード → トランスレーター**を使用して、形式変換の問題をデバッグします。 +Use **Dashboard → Translator** to debug format translation issues: -| モード | いつ使用するか | -| --------------------- | ----------------------------------------------------------------------------------------------------------- | ------------------------ | -| **遊び場** | 入力/出力形式を並べて比較します。失敗したリクエストを貼り付けて、それがどのように変換されるかを確認します。 | -| **チャット テスター** | ライブ メッセージを送信し、ヘッダーを含む完全なリクエスト/レスポンス ペイロードを検査します。 | -| **テストベンチ** | フォーマットの組み合わせ全体でバッチ テストを実行して、どの翻訳が壊れているかを見つけます。 | -| **ライブモニター** | リアルタイムのリクエスト フローを監視して断続的な翻訳の問題を検出 | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**思考タグが表示されない**— ターゲットプロバイダーが思考と思考予算設定をサポートしているかどうかを確認してください -**ツール呼び出しのドロップ**— 一部の形式変換では、サポートされていないフィールドが削除される場合があります。プレイグラウンド モードで確認する -**システム プロンプトがありません**— クロードとジェミニはシステム プロンプトの処理方法が異なります。翻訳出力を確認する -**SDK はオブジェクトではなく生の文字列を返します**— v1.1.0 で修正されました: 応答サニタイザーは、OpenAI SDK Pydantic 検証エラーの原因となる非標準フィールド (`x_groq`、`usage_breakdown` など) を削除するようになりました。-**GLM/ERNIE が「システム」ロールを拒否する**— v1.1.0 で修正: ロール ノーマライザーは、互換性のないモデルのシステム メッセージをユーザー メッセージに自動的にマージします -**`developer` ロールが認識されない**— v1.1.0 で修正: 非 OpenAI プロバイダーの場合は自動的に `system` に変換されます -**`json_schema` が Gemini で動作しない**— v1.1.0 で修正: `response_format` は Gemini の `responseMimeType` + `responseSchema` に変換されるようになりました。--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- 自動レート制限は API キープロバイダーにのみ適用されます (OAuth/サブスクリプションには適用されません) -**設定 → 復元力 → プロバイダー プロファイル**で自動レート制限が有効になっていることを確認します -- プロバイダーが「429」ステータス コードまたは「Retry-After」ヘッダーを返したかどうかを確認します。### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -プロバイダー プロファイルは次の設定をサポートします。 +### Tuning exponential backoff --**基本遅延**— 最初の失敗後の初期待機時間 (デフォルト: 1 秒) -**最大遅延**— 最大待機時間の上限 (デフォルト: 30 秒) -**乗数**— 連続した失敗ごとにどれだけ遅延を増加させるか (デフォルト: 2x)### Anti-thundering herd +Provider profiles support these settings: -多くの同時リクエストがレート制限プロバイダーに到達すると、OmniRoute はミューテックスと自動レート制限を使用してリクエストをシリアル化し、連鎖的な失敗を防ぎます。これは API キープロバイダーの場合は自動的に行われます。--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -一部の OmniRoute ユーザーは、RAG またはエージェント スタックの前にゲートウェイを配置します。これらの設定では、奇妙なパターンがよく見られます。OmniRoute は正常に見えます (プロバイダーは稼働しており、ルーティング プロファイルは正常で、レート制限アラートはありません)。しかし、最終的な答えは依然として間違っています。 +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -実際には、これらのインシデントは通常、ゲートウェイ自体からではなく、ダウンストリームの RAG パイプラインから発生します。 +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -これらの障害を説明するための共有語彙が必要な場合は、16 の繰り返し発生する RAG / LLM 障害パターンを定義する外部 MIT ライセンス テキスト リソースである WFGY 問題マップを使用できます。大まかに説明すると、次のことがカバーされます。 +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- 検索ドリフトと壊れたコンテキスト境界 -- 空または古いインデックスとベクター ストア -- 埋め込みとセマンティックの不一致 -- プロンプトアセンブリとコンテキストウィンドウの問題 -- 論理崩壊と自信過剰な回答 -- 長いチェーンとエージェントの調整の失敗 -- マルチエージェントの記憶と役割のドリフト -- デプロイメントとブートストラップの順序付けの問題 +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -アイデアはシンプルです。 +The idea is simple: -1. 悪い応答を調査する場合は、以下をキャプチャします。 - - ユーザーのタスクとリクエスト - - OmniRoute のルートまたはプロバイダーの組み合わせ - - ダウンストリームで使用される任意の RAG コンテキスト (取得されたドキュメント、ツール呼び出しなど) -2. インシデントを 1 つまたは 2 つの WFGY 問題マップ番号 (「No.1」…「No.16」) にマッピングします。 -3. この番号を独自のダッシュボード、ランブック、またはインシデント トラッカーの OmniRoute ログの隣に保存します。 -4. 対応する WFGY ページを使用して、RAG スタック、レトリーバー、またはルーティング戦略を変更する必要があるかどうかを決定します。 +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -全文と具体的なレシピはここにあります (MIT ライセンス、テキストのみ): +Full text and concrete recipes live here (MIT license, text only): -[WFGY 問題マップの README](https://github.com/onestardao/WFGY/blob/main/問題マップ/README.md) +[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -OmniRoute の背後で RAG またはエージェント パイプラインを実行しない場合は、このセクションを無視してかまいません。--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**GitHub の問題**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**アーキテクチャ**: 内部の詳細については、[`docs/ARCHITECTURE.md`](ARCHITECTURE.md) を参照してください。-**API リファレンス**: すべてのエンドポイントについては、[`docs/API_REFERENCE.md`](API_REFERENCE.md) を参照してください。-**ヘルス ダッシュボード**: リアルタイムのシステム ステータスについては、**ダッシュボード → ヘルス**を確認してください。-**トランスレータ**:**ダッシュボード → トランスレータ**を使用して形式の問題をデバッグします +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/ja/llm.txt b/docs/i18n/ja/llm.txt new file mode 100644 index 0000000000..db2f0ae1a3 --- /dev/null +++ b/docs/i18n/ja/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (日本語) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## 概要 + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### セキュリティ +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/ko/README.md b/docs/i18n/ko/README.md index ef4c69e315..2475f9cf55 100644 --- a/docs/i18n/ko/README.md +++ b/docs/i18n/ko/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_범용 API 프록시 — 하나의 엔드포인트, 60개 이상의 공급자, 가동 중지 시간 없음. 이제**MCP 서버(25개 도구)**,**A2A 프로토콜**,**메모리/스킬 시스템**및**Electron 데스크톱 앱**을 사용할 수 있습니다._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**채팅 완료 • 임베딩 • 이미지 생성 • 비디오 • 음악 • 오디오 • 순위 재지정 •**웹 검색**• MCP 서버 • A2A 프로토콜 • 100% TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _범용 API 프록시 — 하나의 엔드포인트, 60개 이상의 공급자, [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 웹사이트](https://omniroute.online) • [🚀 빠른 시작](#-quick-start) • [💡 기능](#-key-features) • [📖 문서](#-documentation) • [💰 가격](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**사용 가능 언어:**🇺🇸 [영어](README.md) | 🇧🇷 [포르투갈어(브라질)](docs/i18n/pt-BR/README.md) | 🇪🇸 [스페인어](docs/i18n/es/README.md) | 🇫🇷 [프랑스어](docs/i18n/fr/README.md) | 🇮🇹 [이탈리아어](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文(简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [독일어](docs/i18n/de/README.md) | 🇮🇳 [힌디어](docs/i18n/in/README.md) | 🇹🇭 [ไท้](docs/i18n/th/README.md) | 🇺🇦 [Украѕнська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [일본어](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Viet](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [단스크어](docs/i18n/da/README.md) | 🇫🇮 [수오미](docs/i18n/fi/README.md) | 🇮🇱 [언어](docs/i18n/he/README.md) | 🇭🇺 [마자르어](docs/i18n/hu/README.md) | 🇮🇩 [인도네시아어](docs/i18n/id/README.md) | instagram [한국어](docs/i18n/ko/README.md) | 🇲🇾 [바하사 멜라유](docs/i18n/ms/README.md) | 🇳🇱 [네덜란드](docs/i18n/nl/README.md) | 🇳🇴 [노르스크](docs/i18n/no/README.md) | 🇵🇹 [포르투갈어(포르투갈)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [폴스키](docs/i18n/pl/README.md) | 🇸instagram [슬로벤치나](docs/i18n/sk/README.md) | 🇸🇪 [스벤스카](docs/i18n/sv/README.md) | 🇵🇭 [필리핀어](docs/i18n/phi/README.md) | 🇨🇿 [체슈티나](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,555 +60,629 @@ _범용 API 프록시 — 하나의 엔드포인트, 60개 이상의 공급자, ## 📸 Dashboard Preview -<상세> +
+Click to see dashboard screenshots -대시보드 스크린샷을 보려면 클릭하세요. +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | -| 페이지 | 스크린샷 | -| -------------- | ----------------------------------------------- | ---------- | -| **공급자** | ![공급자](docs/screenshots/01-providers.png) | -| **콤보** | ![콤보](docs/screenshots/02-combos.png) | -| **분석** | ![분석](docs/screenshots/03-analytics.png) | -| **건강** | ![건강](docs/screenshots/04-health.png) | -| **번역가** | ![번역기](docs/screenshots/05-translator.png) | -| **설정** | ![설정](docs/screenshots/06-settings.png) | -| **CLI 도구** | ![CLI 도구](docs/screenshots/07-cli-tools.png) | -| **사용 로그** | ![사용법](docs/screenshots/08-usage.png) | -| **엔드포인트** | ![엔드포인트](docs/screenshots/09-endpoint.png) |
| + --- ### 🤖 Free AI Provider for your favorite coding agents -_무제한 코딩을 위한 무료 API 게이트웨이인 OmniRoute를 통해 AI 기반 IDE 또는 CLI 도구를 연결하세요._ - -<테이블> - - - -OpenClaw
-오픈클로 -

-⭐ 205K - - - -NanoBot
-나노봇 -

-⭐ 20.9K - - - -PicoClaw
-피코클로 -

-⭐ 14.6K - - - -ZeroClaw
-제로클로 -

-⭐ 9.9K - - - -IronClaw
-아이언클로 -

-⭐ 2.1K - - - - - -OpenCode
-오픈코드 -

-⭐ 106K - - - -Codex CLI
-코덱스 CLI -

-⭐ 60.8K - - - -클로드 코드
-클로드 코드 -

-⭐ 67.3K - - - -Gemini CLI
-제미니 CLI -

-⭐ 94.7K - - - -킬로 코드
-킬로 코드 -

-⭐ 15.5K - - +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ + + + + + + + + + + + + + + +
+ + OpenClaw
+ OpenClaw +

+ ⭐ 205K +
+ + NanoBot
+ NanoBot +

+ ⭐ 20.9K +
+ + PicoClaw
+ PicoClaw +

+ ⭐ 14.6K +
+ + ZeroClaw
+ ZeroClaw +

+ ⭐ 9.9K +
+ + IronClaw
+ IronClaw +

+ ⭐ 2.1K +
+ + OpenCode
+ OpenCode +

+ ⭐ 106K +
+ + Codex CLI
+ Codex CLI +

+ ⭐ 60.8K +
+ + Claude Code
+ Claude Code +

+ ⭐ 67.3K +
+ + Gemini CLI
+ Gemini CLI +

+ ⭐ 94.7K +
+ + Kilo Code
+ Kilo Code +

+ ⭐ 15.5K +
-📡 모든 에이전트는 http://localhost:20128/v1 또는 http://cloud.omniroute.online/v1를 통해 연결됩니다. 하나의 구성, 무제한 모델 및 할당량--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**돈을 낭비하지 말고 한도에 도달하지 마세요.** +**Stop wasting money and hitting limits:** -- 구독 할당량은 매달 사용하지 않은 채 만료됩니다. -- 속도 제한으로 인해 코딩이 중단됩니다. -- 고가의 API(공급업체당 월 $20-50) -- 공급자 간 수동 전환 +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute는 이 문제를 해결합니다.** +**OmniRoute solves this:** -- ✅**구독 극대화**- 할당량을 추적하고 재설정하기 전에 모든 비트를 사용하세요. -- ✅**자동 대체**- 구독 → API 키 → 저렴한 → 무료, 다운타임 없음 -- ✅**다중 계정**- 공급자별 계정 간 라운드 로빈 -- ✅**유니버설**- Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, 모든 CLI 도구와 함께 작동--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**커뮤니티에 가입하세요!**[WhatsApp 그룹](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — 도움을 받고, 팁을 공유하고, 최신 소식을 받아보세요. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**웹사이트**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**문제**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [커뮤니티 그룹](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**기여**: [CONTRIBUTING.md](CONTRIBUTING.md)를 참조하거나, PR을 열거나, '좋은 첫 호'를 선택하세요. -**원래 프로젝트**: [decolua의 9router](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -이슈를 열 때 system-info 명령을 실행하고 생성된 파일을 첨부하세요.```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -그러면 Node.js 버전, OmniRoute 버전, OS 세부 정보, 설치된 CLI 도구(qoder, gemini, claude, codex, antigravity, droid 등), Docker/PM2 상태, 시스템 패키지 등 문제를 신속하게 재현하는 데 필요한 모든 정보가 포함된 `system-info.txt`가 생성됩니다. GitHub 문제에 직접 파일을 첨부하세요.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**AI 도구를 사용하는 모든 개발자는 매일 이러한 문제에 직면합니다.**OmniRoute는 비용 초과부터 지역적 차단, 손상된 OAuth 흐름부터 프로토콜 운영 및 기업 관찰 가능성까지 모든 문제를 해결하기 위해 구축되었습니다. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. -<상세> -💸 1. "고가의 구독료를 지불했지만 여전히 한도 때문에 방해를 받습니다." +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -개발자는 Claude Pro, Codex Pro 또는 GitHub Copilot에 대해 월 20~200달러를 지불합니다. 비용을 지불하더라도 할당량에는 5시간 사용량, 주간 한도 또는 분당 비율 한도 등 상한선이 있습니다. 코딩 세션이 진행되는 동안 공급자는 응답을 중단하고 개발자는 흐름과 생산성을 잃게 됩니다. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**OmniRoute가 이를 해결하는 방법:** +**How OmniRoute solves it:** --**스마트 4계층 폴백**— 구독 할당량이 소진되면 수동 개입 없이 자동으로 API 키 → 저렴함 → 무료로 리디렉션됩니다. --**공급자 제한 추적**— 캐시된 할당량 스냅샷은 서버 측 일정(기본값 `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`)에 따라 새로 고쳐지며 UI에서 수동 새로 고침이 가능합니다. --**다중 계정 지원**— 자동 라운드 로빈 기능을 갖춘 공급자당 여러 계정 — 하나가 소진되면 다음 계정으로 전환 --**사용자 정의 콤보**— 9가지 균형 전략(우선순위, 가중치 적용, 채우기 우선, 라운드 로빈, P2C, 무작위, 최소 사용, 비용 최적화, 엄격 무작위)을 갖춘 사용자 정의 가능한 폴백 체인 --**Codex 비즈니스 할당량**— 대시보드에서 직접 비즈니스/팀 작업 공간 할당량 모니터링
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard -<상세> -🔌 2. "여러 공급자를 사용해야 하는데 각각 API가 다릅니다." + -OpenAI는 하나의 형식을 사용하고 Claude(Anthropic)는 또 다른 형식을 사용하고 Gemini는 또 다른 형식을 사용합니다. 개발자가 다른 제공업체의 모델을 테스트하거나 이들 사이에서 대체하려는 경우 SDK를 재구성하고, 엔드포인트를 변경하고, 호환되지 않는 형식을 처리해야 합니다. 사용자 지정 공급자(FriendLI, NIM)에는 비표준 모델 끝점이 있습니다. +
+🔌 2. "I need to use multiple providers but each has a different API" -**OmniRoute가 이를 해결하는 방법:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**통합 엔드포인트**— 단일 `http://localhost:20128/v1`이 60개 이상의 모든 공급자에 대한 프록시 역할을 합니다. --**형식 번역**— 자동 및 투명함: OpenAI ← Claude ← Gemini ← Responses API --**응답 삭제**— OpenAI SDK v1.83+를 손상시키는 비표준 필드(`x_groq`, `usage_breakdown`, `service_tier`)를 제거합니다. --**역할 정규화**— OpenAI가 아닌 제공업체에 대해 '개발자' → '시스템'을 변환합니다. GLM/ERNIE의 경우 `시스템` → `사용자` --**Think 태그 추출**— DeepSeek R1과 같은 모델에서 `` 블록을 표준화된 `reasoning_content`로 추출합니다. --**Gemini의 구조화된 출력**— `json_schema` → `responseMimeType`/`responseSchema` 자동 변환 --**`stream`의 기본값은 `false`**— OpenAI 사양에 맞춰 Python/Rust/Go SDK에서 예기치 않은 SSE를 방지합니다.
+**How OmniRoute solves it:** -<상세> -🌐 3. "내 AI 공급자가 내 지역/국가를 차단합니다." +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -OpenAI/Codex와 같은 공급자는 특정 지역의 액세스를 차단합니다. OAuth 및 API 연결 중에 사용자에게 'unsupported_country_region_territory'와 같은 오류가 발생합니다. 이는 특히 개발도상국의 개발자에게 실망스러운 일입니다. + -**OmniRoute가 이를 해결하는 방법:** +
+🌐 3. "My AI provider blocks my region/country" --**3레벨 프록시 구성**— 3가지 레벨로 구성 가능한 프록시: 글로벌(모든 트래픽), 공급자별(하나의 공급자만), 연결/키별 --**색상으로 구분된 프록시 배지**— 시각적 표시기: 🟢 글로벌 프록시, 🟡 공급자 프록시, 🔵 연결 프록시, 항상 IP 표시 --**프록시를 통한 OAuth 토큰 교환**— OAuth 흐름도 프록시를 통과하여 `unsupported_country_region_territory`를 해결합니다. --**프록시를 통한 연결 테스트**- 연결 테스트에서는 구성된 프록시를 사용합니다(더 이상 직접 우회 없음). --**SOCKS5 지원**— 아웃바운드 라우팅을 위한 전체 SOCKS5 프록시 지원 --**TLS 지문 스푸핑**— 'wreq-js'를 통한 브라우저와 유사한 TLS 지문으로 봇 감지 우회 --**🔏 CLI 지문 일치**— 기본 CLI 바이너리 서명과 일치하도록 헤더와 본문 필드의 순서를 변경하여 계정 플래그 위험을 대폭 줄입니다. 프록시 IP는 보존됩니다. 스텔스**및**IP 마스킹을 동시에 얻을 수 있습니다.
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. -<상세> -🆓 4. "AI를 코딩에 활용하고 싶은데 돈이 없어요" +**How OmniRoute solves it:** -모든 사람이 AI 구독 비용으로 월 20~200달러를 지불할 수 있는 것은 아닙니다. 학생, 신흥 국가의 개발자, 취미생활자, 프리랜서는 무료로 고품질 모델에 액세스해야 합니다. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**OmniRoute가 이를 해결하는 방법:** + --**무료 계층 제공자 내장**— 100% 무료 제공자에 대한 기본 지원: Qoder(OAuth를 통한 5개의 무제한 모델: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen(4개의 무제한 모델: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, Vision-model), Kiro (Claude + AWS Builder ID 무료), Gemini CLI(180K 토큰/월 무료) --**Ollama Cloud**— 'api.ollama.com'의 클라우드 호스팅 Ollama 모델(무료 "Light Usage" 계층 포함) `ollamacloud/` 접두사를 사용하세요. --**무료 전용 콤보**— 체인 `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = 다운타임 없이 월 $0 --**NVIDIA NIM 무료 액세스**— ~40RPM 개발 - build.nvidia.com에서 70개 이상의 모델에 영원히 무료 액세스(크레딧에서 순수 속도 제한으로 전환) --**비용 최적화 전략**— 가장 저렴한 공급자를 자동으로 선택하는 라우팅 전략 +
+🆓 4. "I want to use AI for coding but I have no money" -<상세> -🔒 5. "AI 게이트웨이를 무단 액세스로부터 보호해야 합니다." +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -AI 게이트웨이를 네트워크(LAN, VPS, Docker)에 노출하면 주소가 있는 사람은 누구나 개발자의 토큰/할당량을 사용할 수 있습니다. 보호하지 않으면 API는 오용, 즉각적인 주입, 남용에 취약해집니다. +**How OmniRoute solves it:** -**OmniRoute가 이를 해결하는 방법:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**API 키 관리**— 전용 `/dashboard/api-manager` 페이지를 통해 공급자별 생성, 순환 및 범위 지정 --**모델 수준 권한**— 모두 허용/제한 토글을 사용하여 API 키를 특정 모델(`openai/*`, 와일드카드 패턴)로 제한합니다. --**API 엔드포인트 보호**— `/v1/models`에 대한 키를 요구하고 목록에서 특정 공급자를 차단합니다. --**Auth Guard + CSRF 보호**— 'withAuth' 미들웨어 + CSRF 토큰으로 보호되는 모든 대시보드 경로 --**속도 제한기**— 구성 가능한 창으로 IP당 속도 제한 --**IP 필터링**— 액세스 제어를 위한 허용 목록/차단 목록 --**프롬프트 주입 가드**— 악성 프롬프트 패턴 제거 --**AES-256-GCM 암호화**— 저장된 자격 증명은 암호화됩니다.
+ -<상세> -🛑 6. "공급업체가 다운되어 코딩 흐름이 손실되었습니다." +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -AI 제공자는 불안정해지거나, 5xx 오류를 반환하거나, 임시 속도 제한에 도달할 수 있습니다. 개발자가 단일 공급자에 의존하는 경우 중단됩니다. 회로 차단기가 없으면 반복적으로 재시도하면 애플리케이션이 중단될 수 있습니다. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**OmniRoute가 이를 해결하는 방법:** +**How OmniRoute solves it:** --**모델별 회로 차단기**— 구성 가능한 임계값 및 쿨다운(닫힘/열림/반열림)을 통한 자동 열기/닫기, 계단식 블록 방지를 위한 모델별 범위 지정 --**지수 백오프**— 점진적인 재시도 지연 --**Anti-Thundering Herd**— 동시 재시도 폭풍에 대한 뮤텍스 + 세마포어 보호 --**콤보 폴백 체인**— 기본 공급자가 실패하면 개입 없이 자동으로 체인을 통과합니다. --**콤보 회로 차단기**— 콤보 체인 내에서 실패한 공급자를 자동으로 비활성화합니다. --**상태 대시보드**— 가동 시간 모니터링, 회로 차단기 상태, 잠금, 캐시 통계, p50/p95/p99 대기 시간
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest -<상세> -🔧 7. "각각의 AI 도구를 구성하는 것은 지루하고 반복적입니다." + -개발자는 Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code 등을 사용합니다. 각 도구에는 서로 다른 구성(API 엔드포인트, 키, 모델)이 필요합니다. 공급자나 모델을 전환할 때 재구성하는 것은 시간 낭비입니다. +
+🛑 6. "My provider went down and I lost my coding flow" -**OmniRoute가 이를 해결하는 방법:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**CLI 도구 대시보드**— Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline을 원클릭으로 설정할 수 있는 전용 페이지 --**GitHub Copilot Config Generator**— 대량 모델 선택을 통해 VS Code용 `chatLanguageModels.json`을 생성합니다. --**온보딩 마법사**— 처음 사용자를 위한 4단계 설정 안내 --**하나의 엔드포인트, 모든 모델**— `http://localhost:20128/v1`을 한 번 구성하고 60개 이상의 공급자에 액세스
+**How OmniRoute solves it:** -<상세> -🔑 8. "여러 공급자의 OAuth 토큰 관리는 지옥이다" +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot — 모두 만료되는 토큰과 함께 OAuth 2.0을 사용합니다. 개발자는 지속적으로 재인증을 수행하고 'client_secret 누락', 'redirect_uri_mismatch' 및 원격 서버 오류를 처리해야 합니다. LAN/VPS의 OAuth는 특히 문제가 됩니다. + -**OmniRoute가 이를 해결하는 방법:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**자동 토큰 새로 고침**— OAuth 토큰이 만료되기 전에 백그라운드에서 새로 고쳐집니다. --**OAuth 2.0(PKCE) 내장**— Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder에 대한 자동 흐름 --**다중 계정 OAuth**— JWT/ID 토큰 추출을 통해 공급자당 여러 계정 --**OAuth LAN/원격 수정**— `redirect_uri`에 대한 개인 IP 감지 + 원격 서버에 대한 수동 URL 모드 --**Nginx 뒤의 OAuth**— 역방향 프록시 호환성을 위해 `window.location.origin`을 사용합니다. --**원격 OAuth 가이드**— VPS/Docker의 Google Cloud 자격 증명에 대한 단계별 가이드
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. -<상세> -📊 9. "내가 얼마를 어디에 쓰는지 모르겠어요" +**How OmniRoute solves it:** -개발자는 여러 유료 제공업체를 이용하지만 지출에 대한 통합된 보기가 없습니다. 각 제공업체에는 자체 청구 대시보드가 ​​있지만 통합된 보기는 없습니다. 예상치 못한 비용이 쌓일 수 있습니다. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**OmniRoute가 이를 해결하는 방법:** + --**비용 분석 대시보드**— 토큰별 비용 추적 및 공급자별 예산 관리 --**계층당 예산 한도**— 자동 폴백을 트리거하는 계층당 지출 한도 --**모델별 가격 구성**— 모델당 가격 구성 가능 --**API 키당 사용 통계**— 키당 요청 수 및 마지막으로 사용된 타임스탬프 --**분석 대시보드**— 통계 카드, 모델 사용 차트, 성공률 및 대기 시간이 포함된 공급자 테이블 +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" -<상세> -🐛 10. "AI 호출에서는 오류나 문제를 진단할 수 없습니다." +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -호출이 실패하면 개발자는 속도 제한, 만료된 토큰, 잘못된 형식 또는 공급자 오류인지 알 수 없습니다. 여러 터미널에 걸쳐 조각화된 로그. 관찰 가능성이 없으면 디버깅은 시행착오를 겪게 됩니다. +**How OmniRoute solves it:** -**OmniRoute가 이를 해결하는 방법:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**통합 로그 대시보드**— 탭 4개: 요청 로그, 프록시 로그, 감사 로그, 콘솔 --**콘솔 로그 뷰어**— 색상으로 구분된 레벨, 자동 스크롤, 검색, 필터 기능을 갖춘 실시간 터미널 스타일 뷰어 --**SQLite 프록시 로그**— 서버를 다시 시작해도 지속되는 영구 로그 --**번역기 플레이그라운드**— 4가지 디버깅 모드: 플레이그라운드(형식 번역), 채팅 테스터(왕복), 테스트 벤치(일괄), 라이브 모니터(실시간) --**원격 측정 요청**— p50/p95/p99 대기 시간 + X-요청-ID 추적 --**회전을 통한 파일 기반 로깅**— 앱 로그는 크기, 보존 일수 및 아카이브 수에 따라 회전합니다. 통화 로그 아티팩트는 보존 일수 및 파일 수에 따라 순환됩니다. --**시스템 정보 보고서**— `npm run system-info`는 전체 환경(노드 버전, OmniRoute 버전, OS, CLI 도구, Docker/PM2 상태)과 함께 `system-info.txt`를 생성합니다. 즉각적인 분류를 위해 문제를 보고할 때 첨부하세요.
+ -<상세> -🏗️ 11. "게이트웨이 배포 및 유지 관리가 복잡합니다." +
+📊 9. "I don't know how much I'm spending or where" -다양한 환경(로컬, VPS, Docker, 클라우드)에서 AI 프록시를 설치, 구성 및 유지 관리하는 것은 노동 집약적입니다. 하드코딩된 경로, 디렉터리의 'EACCES', 포트 충돌, 크로스 플랫폼 빌드와 같은 문제로 인해 마찰이 가중됩니다. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**OmniRoute가 이를 해결하는 방법:** +**How OmniRoute solves it:** --**npm 전역 설치**— `npm install -g omniroute && omniroute` — 완료 --**Docker 다중 플랫폼**— AMD64 + ARM64 기본(Apple Silicon, AWS Graviton, Raspberry Pi) --**Docker Compose 프로필**— `base`(CLI 도구 없음) 및 `cli`(Claude Code, Codex, OpenClaw 포함) --**Electron 데스크탑 앱**— 시스템 트레이, 자동 시작, 오프라인 모드를 갖춘 Windows/macOS/Linux용 기본 앱 --**분할 포트 모드**— 고급 시나리오(역방향 프록시, 컨테이너 네트워킹)를 위한 별도의 포트에 있는 API 및 대시보드 --**클라우드 동기화**— Cloudflare Workers를 통해 장치 간 구성 동기화 --**DB 백업**— 외부에서 관리되는 백업의 경우 `DISABLE_SQLITE_AUTO_BACKUP`을 사용하여 모든 설정의 자동 백업, 복원, 내보내기 및 가져오기
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency -<상세> -🌍 12. "인터페이스는 영어로만 제공되며 우리 팀은 영어를 사용하지 않습니다." + -영어를 사용하지 않는 국가, 특히 라틴 아메리카, 아시아, 유럽의 팀은 영어 전용 인터페이스로 인해 어려움을 겪고 있습니다. 언어 장벽으로 인해 채택이 줄어들고 구성 오류가 증가합니다. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**OmniRoute가 이를 해결하는 방법:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**대시보드 i18n — 30개 언어**— 한국어, 아랍어, 불가리아어, 덴마크어, 독일어, 스페인어, 핀란드어, 프랑스어, 히브리어, 힌디어, 헝가리어, 인도네시아어, 이탈리아어, 일본어, 말레이어, 네덜란드어, 노르웨이어, 폴란드어, 포르투갈어(PT/BR), 루마니아어, 러시아어, 슬로바키아어, 스웨덴어, 태국어, 우크라이나어, 베트남어, 중국어, 필리핀어, 영어를 포함한 500개 이상의 키 번역됨 --**RTL 지원**— 아랍어 및 히브리어에 대해 오른쪽에서 왼쪽으로 지원 --**다국어 README**— 30개의 완전한 문서 번역 --**언어 선택기**— 실시간 전환을 위한 헤더의 지구본 아이콘
+**How OmniRoute solves it:** -<상세> -🔄 13. "채팅 이상의 것이 필요합니다. 임베딩, 이미지, 오디오가 필요합니다." +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -AI는 단순한 채팅 완성이 아닙니다. 개발자는 이미지를 생성하고, 오디오를 기록하고, RAG용 임베딩을 만들고, 문서 순위를 다시 지정하고, 콘텐츠를 조정해야 합니다. 각 API에는 서로 다른 엔드포인트와 형식이 있습니다. + -**OmniRoute가 이를 해결하는 방법:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**임베딩**— 6개 공급자와 9개 이상의 모델이 포함된 `/v1/embeddings` --**이미지 생성**— 10개 공급자와 20개 이상의 모델(OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI)이 포함된 `/v1/images/세대` --**텍스트-비디오**— `/v1/videos/세대` — ComfyUI(AnimateDiff, SVD) 및 SD WebUI --**텍스트-음악**— `/v1/music/세대` — ComfyUI(안정적인 오디오 오픈, MusicGen) --**오디오 전사**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**텍스트 음성 변환**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + 기존 제공업체 --**조정**— `/v1/moderations` — 콘텐츠 안전 확인 --**재순위**— `/v1/rerank` — 문서 관련성 재순위 --**응답 API**— Codex에 대한 전체 `/v1/responses` 지원
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. -<상세> -🧪 14. "모델별로 품질을 테스트하고 비교할 방법이 없습니다." +**How OmniRoute solves it:** -개발자는 자신의 사용 사례(코드, 번역, 추론)에 가장 적합한 모델을 알고 싶어하지만 수동으로 비교하는 것은 느립니다. 통합 평가 도구가 없습니다. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**OmniRoute가 이를 해결하는 방법:** + --**LLM 평가**— 인사말, 수학, 지리, 코드 생성, JSON 준수, 번역, 마크다운, 안전 거부를 다루는 사전 로드된 10가지 사례를 사용한 골든 세트 테스트 --**4가지 일치 전략**— `exact`, `contains`, `regex`, `custom`(JS 함수) --**번역기 플레이그라운드 테스트 벤치**— 여러 입력 및 예상 출력을 사용한 일괄 테스트, 공급업체 간 비교 --**채팅 테스터**— 시각적 응답 렌더링을 포함한 전체 왕복 --**라이브 모니터**— 프록시를 통해 흐르는 모든 요청의 실시간 스트림 +
+🌍 12. "The interface is English-only and my team doesn't speak English" -<상세> -📈 15. "성능 저하 없이 확장해야 합니다" +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -요청량이 증가함에 따라 동일한 질문을 캐싱하지 않으면 중복 비용이 발생합니다. 멱등성이 없으면 중복 요청으로 인해 처리가 낭비됩니다. 공급자별 요금 제한을 준수해야 합니다. +**How OmniRoute solves it:** -**OmniRoute가 이를 해결하는 방법:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**의미 체계 캐시**— 2계층 캐시(서명 + 의미 체계)로 비용과 대기 시간 감소 --**멱등성 요청**— 동일한 요청에 대한 중복 제거 기간은 5초입니다. --**속도 제한 감지**— 제공업체별 RPM, 최소 간격 및 최대 동시 추적 --**편집 가능한 속도 제한**— 설정 → 지속성을 통한 복원력에서 구성 가능한 기본값 --**API 키 검증 캐시**— 프로덕션 성능을 위한 3계층 캐시 --**원격 측정 기능을 갖춘 상태 대시보드**— p50/p95/p99 대기 시간, 캐시 통계, 가동 시간
+ -<상세> -🤖 16. "모델 동작을 전체적으로 제어하고 싶습니다." +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -특정 언어, 특정 어조로 모든 응답을 원하거나 추론 토큰을 제한하려는 개발자. 모든 도구/요청에서 이를 구성하는 것은 비현실적입니다. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**OmniRoute가 이를 해결하는 방법:** +**How OmniRoute solves it:** --**시스템 프롬프트 삽입**— 모든 요청에 전역 프롬프트 적용 --**생각 예산 검증**— 요청별 토큰 할당 제어 추론(패스스루, 자동, 사용자 정의, 적응형) --**9가지 라우팅 전략**— 요청 배포 방법을 결정하는 글로벌 전략 --**와일드카드 라우터**— `provider/*` 패턴은 모든 공급자에게 동적으로 라우팅됩니다. --**콤보 활성화/비활성화 토글**— 대시보드에서 직접 콤보를 토글합니다. --**공급자 토글**— 한 번의 클릭으로 공급자에 대한 모든 연결을 활성화/비활성화합니다. --**차단된 제공자**— `/v1/models` 목록에서 특정 제공자를 제외합니다.
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex -<상세> -🧰 17. "최고급 제품 기능으로 MCP 도구가 필요합니다" + -많은 AI 게이트웨이는 MCP를 숨겨진 구현 세부 사항으로만 노출합니다. 팀에는 가시적이고 관리 가능한 운영 레이어가 필요합니다. +
+🧪 14. "I have no way to test and compare quality across models" -**OmniRoute가 이를 해결하는 방법:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- 대시보드 탐색 및 엔드포인트 프로토콜 탭에 MCP가 나타납니다. -- 프로세스, 도구, 범위 및 감사가 포함된 전용 MCP 관리 페이지 -- `omniroute --mcp` 및 클라이언트 온보딩을 위한 기본 제공 빠른 시작
+**How OmniRoute solves it:** -<상세> -🧠 18. "동기화 + 스트림 작업 경로를 갖춘 A2A 오케스트레이션이 필요합니다." +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -에이전트 워크플로에는 수명 주기 제어를 통해 직접 응답과 장기 실행 스트리밍 실행이 모두 필요합니다. + -**OmniRoute가 이를 해결하는 방법:** +
+📈 15. "I need to scale without losing performance" -- `message/send` 및 `message/stream`이 포함된 A2A JSON-RPC 엔드포인트(`POST /a2a`) -- 터미널 상태 전파를 통한 SSE 스트리밍 -- `tasks/get` 및 `tasks/cancel`을 위한 작업 수명 주기 API
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. -<상세> -🛰️ 19. "추측된 상태가 아닌 실제 MCP 프로세스 상태가 필요합니다." +**How OmniRoute solves it:** -운영팀은 API에 접근할 수 있는지 여부뿐만 아니라 MCP가 실제로 활성화되어 있는지 알아야 합니다. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**OmniRoute가 이를 해결하는 방법:** + -- PID, 타임스탬프, 전송, 도구 개수 및 범위 모드가 포함된 런타임 하트비트 파일 -- 하트비트 + 최근 활동을 결합한 MCP 상태 API -- 프로세스/가동 시간/하트비트 최신성을 위한 UI 상태 카드 +
+🤖 16. "I want to control model behavior globally" -<상세> -📋 20. "감사 가능한 MCP 도구 실행이 필요합니다." +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -도구가 구성을 변경하거나 운영 작업을 트리거하는 경우 팀에는 법의학적 추적성이 필요합니다. +**How OmniRoute solves it:** -**OmniRoute가 이를 해결하는 방법:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- MCP 도구 호출에 대한 SQLite 지원 감사 로깅 -- 도구, 성공/실패, API 키, 페이지 매김 기준으로 필터링 -- 자동화를 위한 대시보드 감사 테이블 + 통계 엔드포인트
+ -<상세> -🔐 21. "통합별로 범위가 지정된 MCP 권한이 필요합니다." +
+🧰 17. "I need MCP tools as first-class product capabilities" -다양한 클라이언트에는 도구 범주에 대한 최소 권한 액세스 권한이 있어야 합니다. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**OmniRoute가 이를 해결하는 방법:** +**How OmniRoute solves it:** -- 제어된 도구 액세스를 위한 10개의 세분화된 MCP 범위 -- MCP 관리 UI의 범위 적용 및 가시성 -- 운영 툴링을 위한 안전한 기본 자세
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding -<상세> -⚙️ 22. "재배치 없이 운영 통제가 필요해요" + -팀은 사고 또는 비용 이벤트 중에 빠른 런타임 변경이 필요합니다. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**OmniRoute가 이를 해결하는 방법:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- MCP 대시보드에서 직접 콤보 활성화 전환 -- 사전 정의된 정책 팩의 복원력 프로필 적용 -- 동일한 운영 패널에서 회로 차단기 상태 재설정
+**How OmniRoute solves it:** -<상세> -🔄 23. "실시간 A2A 작업 수명주기 가시성 및 취소가 필요합니다." +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -수명주기 가시성이 없으면 작업 사고를 분류하기가 어려워집니다. + -**OmniRoute가 이를 해결하는 방법:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- 페이지 매김을 통한 상태/기술별 작업 목록/필터링 -- 작업 메타데이터, 이벤트 및 아티팩트에 대한 드릴다운 -- 작업 취소 끝점 및 확인이 포함된 UI 작업
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. -<상세> -🌊 24. "A2A 로드를 위한 활성 스트림 측정항목이 필요합니다." +**How OmniRoute solves it:** -스트리밍 워크플로에는 동시성 및 라이브 연결에 대한 운영 통찰력이 필요합니다. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**OmniRoute가 이를 해결하는 방법:** + -- A2A 상태에 통합된 활성 스트림 카운터 -- 마지막 작업 타임스탬프 및 상태별 개수 -- 실시간 운영 모니터링을 위한 A2A 대시보드 카드 +
+📋 20. "I need auditable MCP tool execution" -<상세> -🪪 25. "클라이언트를 위한 표준 에이전트 검색이 필요합니다." +When tools mutate config or trigger ops actions, teams need forensic traceability. -외부 클라이언트 및 오케스트레이터에는 온보딩을 위해 컴퓨터에서 읽을 수 있는 메타데이터가 필요합니다. +**How OmniRoute solves it:** -**OmniRoute가 이를 해결하는 방법:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- `/.well-known/agent.json`에 에이전트 카드가 노출됨 -- 관리 UI에 표시되는 능력과 기술 -- A2A 상태 API에는 자동화를 위한 검색 메타데이터가 포함되어 있습니다.
+ -<상세> -🧭 26. "제품 UX에서 프로토콜 검색 기능이 필요합니다." +
+🔐 21. "I need scoped MCP permissions per integration" -사용자가 프로토콜 표면을 발견할 수 없는 경우 채택 및 지원 품질이 저하됩니다. +Different clients should have least-privilege access to tool categories. -**OmniRoute가 이를 해결하는 방법:** +**How OmniRoute solves it:** -- 프록시, MCP, A2A 및 API 엔드포인트에 대한 탭이 포함된 통합**엔드포인트**페이지 -- MCP 및 A2A에 대한 인라인 서비스 상태 토글(온라인/오프라인) -- 개요에서 전용 관리 탭으로의 링크
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling -<상세> -🧪 27. "실제 클라이언트와의 엔드투엔드 프로토콜 검증이 필요합니다" + -모의 테스트는 출시 전에 프로토콜 호환성을 검증하기에 충분하지 않습니다. +
+⚙️ 22. "I need operational controls without redeploying" -**OmniRoute가 이를 해결하는 방법:** +Teams need quick runtime changes during incidents or cost events. -- 앱을 부팅하고 실제 MCP SDK 클라이언트 전송을 사용하는 E2E 제품군 -- 흐름 검색, 전송, 스트리밍, 가져오기 및 취소에 대한 A2A 클라이언트 테스트 -- MCP 감사 및 A2A 작업 API에 대한 교차 확인 주장
+**How OmniRoute solves it:** -<상세> -📡 28. "모든 인터페이스에 걸쳐 통합된 관찰 가능성이 필요합니다." +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -프로토콜별로 관찰 가능성을 분할하면 사각지대가 발생하고 MTTR이 길어집니다. + -**OmniRoute가 이를 해결하는 방법:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- 대시보드/로그/분석을 하나의 제품으로 통합 -- OpenAI, MCP 및 A2A 계층 전반에 걸쳐 상태 + 감사 + 원격 측정 요청 -- 상태 및 자동화를 위한 운영 API
+Without lifecycle visibility, task incidents become hard to triage. -<상세> -💼 29. "프록시 + 도구 + 에이전트 오케스트레이션을 위해 하나의 런타임이 필요합니다." +**How OmniRoute solves it:** -여러 개별 서비스를 실행하면 운영 비용과 오류 모드가 증가합니다. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**OmniRoute가 이를 해결하는 방법:** + -- OpenAI 호환 프록시, MCP 서버, A2A 서버가 하나의 스택에 있음 -- 공유 인증, 복원력, 데이터 저장소 및 관찰 가능성 -- 모든 상호 작용 표면에 걸쳐 일관된 정책 모델 +
+🌊 24. "I need active stream metrics for A2A load" -<상세> -🚀 30. "글루 코드 확장 없이 에이전트 워크플로를 제공해야 합니다." +Streaming workflows require operational insight into concurrency and live connections. -여러 임시 서비스와 스크립트를 결합할 때 팀의 속도가 느려집니다. +**How OmniRoute solves it:** -**OmniRoute가 이를 해결하는 방법:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- 클라이언트와 에이전트를 위한 통합 엔드포인트 전략 -- 내장된 프로토콜 관리 UI 및 연기 검증 경로 -- 프로덕션 준비 기반(보안, 로깅, 탄력성, 백업)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**플레이북 A: 유료 구독 극대화 + 저렴한 백업**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -609,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**플레이북 B: 비용이 전혀 들지 않는 코딩 스택**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**플레이북 C: 연중무휴 상시 가동 폴백 체인**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -632,125 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**플레이북 D: MCP + A2A를 사용한 에이전트 작업**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost ->**$0/월**로 몇 분 만에 AI 코딩을 설정할 수 있습니다. 무료 계정을 연결하고 내장된**무료 스택**콤보를 사용하세요. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| 단계 | 액션 | 제공자 잠금 해제 | -| ---- | ------------------------------------- | ----------------------------------------------------- | -| 1 |**Kiro**연결(AWS Builder ID OAuth) | 클로드 소네트 4.5, 하이쿠 4.5 —**무제한**| -| 2 |**Qoder**연결(Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... —**무제한**| -| 3 | 연결**Qwen**(장치 코드) | qwen3-coder-plus, qwen3-coder-flash... —**무제한**| -| 4 |**Gemini CLI**(Google OAuth) 연결 | gemini-3-flash, gemini-2.5-pro —**180K/월 무료**| -| 5 | `/dashboard/combos` →**Free Stack ($0)**템플릿 | 모든 무료 공급자를 자동으로 라운드 로빈 | +| Step | Action | Providers Unlocked | +| ---- | -------------------------------------------------- | ------------------------------------------------------------------ | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**모든 IDE/CLI를 다음으로 지정하세요:**`http://localhost:20128/v1` · API 키: `any-string` · 완료. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**선택적 추가 적용 범위(무료):**Groq API 키(30RPM 무료), NVIDIA NIM(40RPM 무료, 70개 이상의 모델), Cerebras(100만 토크/일), LongCat API 키(5000만 토큰/일!), Cloudflare Workers AI(10K 뉴런/일, 50개 이상의 모델).## 빠른 시작 +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## 빠른 시작 ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **pnpm 사용자:**`better-sqlite3` 및 `@swc/core`에 필요한 기본 빌드 스크립트를 활성화하려면 설치 후 `pnpm recognition-builds -g`를 실행하세요. +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > -> ``배쉬 +> ```bash > pnpm install -g omniroute -> pnpm recognition-builds -g # 모든 패키지 선택 → 승인 -> 옴니루트 -> -> ``` -> +> pnpm approve-builds -g # Select all packages → approve +> omniroute > ``` -대시보드는 `http://localhost:20128`에서 열리고 API 기본 URL은 `http://localhost:20128/v1`입니다. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| 명령 | 설명 | -| ----------------------- | ------------------------------------------------------ | -| '옴니루트' | 서버 시작(`PORT=20128`, 동일한 포트에 API 및 대시보드) | -| `omniroute --port 3000` | 표준/API 포트를 3000으로 설정 | -| `옴니루트 --mcp` | MCP 서버 시작(stdio 전송) | -| `omniroute --no-open` | 브라우저를 자동으로 열지 않음 | -| `omniroute --help` | 도움말 표시 | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -선택적 분할 포트 모드:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -대부분의 배포에는 다음만 필요합니다. +For most deployments, you only need: -| 변수 | 기본값 | 목적 | -| ----------- | ---------------- | ------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600000` | 업스트림 가져오기, 숨겨진 Undici 시간 초과, TLS 지문 요청 및 API 브리지 요청/프록시 시간 초과에 대한 공유 기준 | -| `STREAM_IDLE_TIMEOUT_MS` | REQUEST_TIMEOUT_MS`를 상속합니다 | OmniRoute가 SSE 스트림을 중단하기 전 스트리밍 청크 간의 최대 간격 | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -이전 버전과의 호환성은 유지됩니다. 기존 `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` 및 기타 레이어별 시간 제한 변수는 계속 작동하며 공유 기준을 재정의합니다. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -더 세밀한 제어가 필요한 경우 고급 재정의를 사용할 수 있습니다.| Variable | 기본값 | 목적 | -| --------------------------- | ----------------------------- | ------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | REQUEST_TIMEOUT_MS`를 상속합니다 | 기본 가져오기 중단 신호에 사용되는 총 업스트림 요청 시간 초과 | -| `FETCH_HEADERS_TIMEOUT_MS` | 'FETCH_TIMEOUT_MS'를 상속합니다 | 업스트림 응답 헤더 수신을 위한 Undici 시간 제한 | -| `FETCH_BODY_TIMEOUT_MS` | 'FETCH_TIMEOUT_MS'를 상속합니다 | 업스트림 본문 청크 사이의 Undici 시간 제한(`0`은 비활성화) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP 연결 시간 초과 | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | '4000' | Undici 유휴 연결 유지 소켓 시간 초과 | -| `TLS_CLIENT_TIMEOUT_MS` | 'FETCH_TIMEOUT_MS'를 상속합니다 | `wreq-js`를 통해 수행된 TLS 지문 요청 시간 초과 | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | REQUEST_TIMEOUT_MS` 또는 `30000`을 상속합니다 | API 포트에서 대시보드 포트로의 `/v1` 프록시 전달 시간 초과 | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `최대(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | API 브릿지 서버의 수신 요청 시간 초과 | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | API 브릿지 서버의 수신 헤더 시간 초과 | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | API 브릿지 서버의 연결 유지 시간 초과 | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | API 브릿지 서버의 소켓 비활성 시간 초과(`0`은 비활성화) | +Advanced overrides are available if you need finer control: -Nginx, Caddy, Cloudflare 또는 다른 역방향 프록시 뒤에서 OmniRoute를 실행하는 경우 프록시가 -시간 제한은 OmniRoute 스트림/가져오기 시간 제한보다 높습니다.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. 대시보드 → `공급자`를 열고 하나 이상의 공급자(OAuth 또는 API 키)를 연결합니다. -2. 대시보드 → `Endpoints`를 열고 API 키를 생성합니다. -3. (선택 사항) 대시보드 → `콤보`를 열고 폴백 체인을 설정합니다.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode 및 OpenAI 호환 SDK와 함께 작동합니다.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP(도구 기반 작업용):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` +Then connect your MCP client over `stdio` and test tools like: -그런 다음 `stdio`를 통해 MCP 클라이언트를 연결하고 다음과 같은 테스트 도구를 사용하세요. +- `omniroute_get_health` +- `omniroute_list_combos` --`omniroute_get_health` --`omniroute_list_combos` +**A2A (for agent-to-agent workflows):** -**A2A(에이전트 간 워크플로용):**```bash +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -764,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -이 제품군은 실행 중인 앱에 대해 실제 MCP 및 A2A 클라이언트 흐름의 유효성을 검사합니다.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -772,14 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` -<상세> +
+Void Linux (`xbps-src` template) -Void Linux(`xbps-src` 템플릿) - -Void Linux 사용자의 경우 `xbps-src`를 사용하여 기본 패키지를 빌드할 수 있습니다. 이 블록을 `srcpkgs/omniroute/template`으로 저장합니다.```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -791,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -799,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -875,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -886,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute는 [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute)에서 공개 Docker 이미지로 제공됩니다. +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**빠른 실행:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -896,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**환경 파일 포함:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Docker Compose 사용:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Docker 배포를 위한 대시보드 지원에는 이제 '대시보드 → 엔드포인트'에서 원클릭**Cloudflare Quick Tunnel**이 포함됩니다. 첫 번째 활성화는 필요한 경우에만 `cloudflared`를 다운로드하고 현재 `/v1` 끝점에 대한 임시 터널을 시작하며 생성된 `https://*.trycloudflare.com/v1` URL을 일반 공개 URL 바로 아래에 표시합니다. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -참고: +Notes: -- 빠른 터널 URL은 일시적이며 다시 시작할 때마다 변경됩니다. -- 빠른 터널은 OmniRoute 또는 컨테이너를 다시 시작한 후에 자동으로 복원되지 않습니다. 필요할 때 대시보드에서 다시 활성화하세요. -- 관리형 설치는 현재 `x64`/`arm64`에서 Linux, macOS 및 Windows를 지원합니다. -- 제한된 컨테이너 환경에서 시끄러운 QUIC UDP 버퍼 경고를 방지하기 위해 관리형 빠른 터널은 기본적으로 HTTP/2 전송으로 설정됩니다. 다른 전송을 원할 경우 `CLOUDFLARED_PROTOCOL=quic` 또는 `auto`를 설정하세요. -- Docker 이미지는 시스템 CA 루트를 번들로 묶어 관리형 'cloudflared'에 전달합니다. 이는 터널이 컨테이너 내부에서 부트스트랩할 때 TLS 신뢰 실패를 방지합니다. -- SQLite는 WAL 모드에서 실행됩니다. OmniRoute가 최신 변경 사항을 다시 `storage.sqlite`로 검사할 수 있도록 `docker stop`이 완료되도록 허용해야 합니다. -- 번들 Compose 파일에는 이미 40초 중지 유예 기간이 설정되어 있습니다. 이미지를 직접 실행하는 경우 `--stop-timeout 40`(또는 유사)을 유지하여 수동 중지로 인해 종료 정리가 중단되지 않도록 하세요. -- OmniRoute가 다운로드하는 대신 기존 바이너리를 사용하도록 하려면 `CLOUDFLARED_BIN=/absolute/path/to/cloudflared`를 설정합니다. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Caddy와 함께 Docker Compose 사용(HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute는 Caddy의 자동 SSL 프로비저닝을 사용하여 안전하게 노출될 수 있습니다. 도메인의 DNS A 레코드가 서버의 IP를 가리키는지 확인하세요.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` +| Image | Tag | Size | Description | +| ------------------------ | -------- | ------ | --------------------- | +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | -| 이미지 | 태그 | 사이즈 | 설명 | -| ----------- | -------- | ------ | -------- | -| `diegosouzapw/omniroute` | `최신` | ~250MB | 최신 안정 릴리스 | -| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | 현재 버전 |--- +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**신규!**이제 OmniRoute를 Windows, macOS, Linux용**기본 데스크톱 애플리케이션**으로 사용할 수 있습니다. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -OmniRoute를 독립형 데스크톱 앱으로 실행하세요. 로컬 모델에는 터미널, 브라우저, 인터넷이 필요하지 않습니다. Electron 기반 앱에는 다음이 포함됩니다. +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**기본 창**— 시스템 트레이 통합이 가능한 전용 앱 창 -- 🔄**자동 시작**— 시스템 로그인 시 OmniRoute 실행 -- 🔔**기본 알림**— 할당량 소진 또는 공급자 문제에 대한 알림 받기 -- ⚡**원클릭 설치**— NSIS(Windows), DMG(macOS), AppImage(Linux) -- 🌐**오프라인 모드**— 번들 서버를 사용하여 완전히 오프라인으로 작동### 빠른 시작 +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### 빠른 시작 ```bash # Development mode @@ -985,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -최소화되면 OmniRoute는 빠른 작업을 통해 시스템 트레이에 표시됩니다. +When minimized, OmniRoute lives in your system tray with quick actions: -- 대시보드 열기 -- 서버 포트 변경 -- 애플리케이션 종료 +- Open dashboard +- Change server port +- Quit application -📖 전체 문서: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| 계층 | 공급자 | 비용 | 할당량 재설정 | 최고의 대상 | -| ------------- | ----------------------- | ------------------------------ | --------------- | -------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 구독** | 클로드 코드 (Pro) | $20/월 | 5시간 + 매주 | 이미 구독 중 | -| | 코덱스(플러스/프로) | $20-200/월 | 5시간 + 매주 | OpenAI 사용자 | -| | 제미니 CLI | **무료** | 180K/월 + 1K/일 | 모든 사람! | -| | GitHub 부조종사 | $10-19/월 | 월간 | GitHub 사용자 | -| **🔑 API 키** | 엔비디아 NIM | **무료**(영원한 개발) | ~40RPM | 70개 이상의 공개 모델 | -| | 대뇌 | **무료**(100만 톡/일) | 60K TPM/30RPM | 세계에서 가장 빠른 | -| | 그로크 | **무료**(30RPM) | 14.4KRPD | 초고속 라마/젬마 | -| | DeepSeek V3.2 | 100만 달러당 $0.27/$1.10 | 없음 | 최고의 가격/품질 추론 | -| | xAI Grok-4 고속 | **100만 달러당 $0.20/$0.50**🆕 | 없음 | 가장 빠른 + 도구 호출, 초저 | -| | xAI Grok-4(표준) | 100만 달러당 $0.20/$1.50 🆕 | 없음 | xAI의 추론 주력 | -| | 미스트랄 | 무료 평가판 + 유료 | 요금 제한 | 유럽의 AI | -| | 오픈라우터 | 종량제 | 없음 | 100개 이상의 모델이 결합되어 있습니다. | -| **💰 저렴한** | GLM-5(Z.AI를 통해) 🆕 | $0.5/1M | 매일 오전 10시 | 128K 출력, 최신 플래그십 | -| | GLM-4.7 | $0.6/1M | 매일 오전 10시 | 예산 백업 | -| | 미니맥스 M2.5 🆕 | $0.3/1M 입력 | 5시간 롤링 | 추론 + 에이전트 작업 | -| | 미니맥스 M2.1 | $0.2/1M | 5시간 롤링 | 가장 저렴한 옵션 | -| | Kimi K2.5(문샷 API) 🆕 | 종량제 | 없음 | 직접 Moonshot API 액세스 | -| | 키미 K2 | $9/월 정액 | 1000만 토큰/월 | 예측 가능한 비용 | -| **🆓 무료** | Qoder | **$0** | 무제한 | 5개 모델 무제한 | -| | 퀀 | **$0** | 무제한 | 4개 모델 무제한 | -| | 키로 | **$0** | 무제한 | 클로드 소네트/하이쿠(AWS 빌더) | -| | LongCat 플래시라이트 🆕 | **$0**(5천만 토크/일 🔥) | 1RPS | 지구상에서 가장 큰 무료 할당량 | -| | 수분 AI 🆕 | **$0**(열쇠 필요 없음) | 요청 1개/15초 | GPT-5, 클로드, DeepSeek, 라마 4 | -| | Cloudflare 작업자 AI 🆕 | **$0**(10,000개의 뉴런/일) | ~150회/일 | 50개 이상의 모델, 글로벌 엣지 | -| | 스케일웨이 AI 🆕 | **$0**(총 1백만 토큰) | 요금 제한 | EU/GDPR, Qwen3 235B, 라마 70B | > 🆕**새 모델 추가됨(2026년 3월):**$0.20/$0.50/M의 Grok-4 Fast 제품군(1143ms로 벤치마크 — Gemini 2.5 Flash보다 30% 빠름), 128K 출력의 Z.AI를 통한 GLM-5, MiniMax M2.5 추론, DeepSeek V3.2 업데이트된 가격, Moonshot 직접 API를 통한 Kimi K2.5. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 $0 콤보 스택 — 완전한 무료 설정:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -````` - -**비용이 0입니다. 코딩을 중단하지 마세요.**이를 하나의 OmniRoute 콤보로 구성하면 모든 폴백이 자동으로 발생하므로 수동 전환이 필요하지 않습니다.--- +--- --- ## 🆓 Free Models — What You Actually Get -> 아래의 모든 모델은**신용카드가 필요 없이 100% 무료**입니다. 하나의 할당량이 소진되면 OmniRoute는 둘 사이를 자동으로 라우팅합니다. 이를 모두 결합하여 깨지지 않는 $0 콤보를 만듭니다.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| 모델 | 접두사 | 한도 | 비율 제한 | -| ------ | ------ | ------------- | -------- | -| `클로드-소네트-4.5` | `kr/` |**무제한**| 보고된 일일 한도 없음 | -| `claude-haiku-4.5` | `kr/` |**무제한**| 보고된 일일 한도 없음 | -| `클로드-오푸스-4.6` | `kr/` |**무제한**| Kiro를 통한 최신 Opus |### 🟢 QODER MODELS (Free PAT via qodercli) +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) -| 모델 | 접두사 | 한도 | 비율 제한 | +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | --------------------- | +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | + +### 🟢 QODER MODELS (Free PAT via qodercli) + +| Model | Prefix | Limit | Rate Limit | | ------------------ | ------ | ------------- | --------------- | -| `kimi-k2-생각` | `만약/` |**무제한**| 보고된 한도 없음 | -| `qwen3-coder-plus` | `만약/` |**무제한**| 보고된 한도 없음 | -| '깊은 탐색-r1' | `만약/` |**무제한**| 보고된 한도 없음 | -| `minimax-m2.1` | `만약/` |**무제한**| 보고된 한도 없음 | -| '키미-k2' | `만약/` |**무제한**| 보고된 한도 없음 | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -> 권장 연결 방법:**Personal Access Token + `qodercli`**. 브라우저 OAuth는 -> 'QODER_OAUTH_*' 환경 변수가 구성되지 않은 한 실험적이며 기본적으로 비활성화됩니다.### 🟡 QWEN MODELS (Device Code Auth) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| 모델 | 접두사 | 한도 | 비율 제한 | -| ------ | ------ | ------------- | ------ | -| `qwen3-coder-plus` | `qw/` |**무제한**| 보고된 한도 없음 | -| `qwen3-coder-flash` | `qw/` |**무제한**| 보고된 한도 없음 | -| `qwen3-coder-next` | `qw/` |**무제한**| 보고된 한도 없음 | -| `비전 모델` | `qw/` |**무제한**| 다중 모드(이미지) |### 🟣 GEMINI CLI (Google OAuth) +### 🟡 QWEN MODELS (Device Code Auth) -| 모델 | 접두사 | 한도 | 비율 제한 | -| ----------- | ------ | -------------- | ------------- | -| `gemini-3-플래시 미리보기` | `gc/` |**180K 톡/월**+ 1K/일 | 월별 재설정 | -| `gemini-2.5-pro` | `gc/` | 180K/월(공유 풀) | 고품질 |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | ------------------- | +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| 계층 | 일일 한도 | 비율 제한 | 메모 | -| ---------- | ------------ | ----------- | ----------------------------------------- | -| 무료(개발자) | 토큰 한도 없음 |**~40RPM**| 70개 이상의 모델; 2025년 중반 순수 비율 제한으로 전환 | +### 🟣 GEMINI CLI (Google OAuth) -인기 무료 모델: `moonshotai/kimi-k2.5`(Kimi K2.5), `z-ai/glm4.7`(GLM 4.7), `deepseek-ai/deepseek-v3.2`(DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -| 계층 | 일일 한도 | 비율 제한 | 메모 | -| ---- | ----------------- | ---------------- | ------------------------------ | -| 무료 |**100만 토큰/일**| 60K TPM/30RPM | 세계에서 가장 빠른 LLM 추론; 매일 재설정 | +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) -무료로 사용 가능: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +| Tier | Daily Limit | Rate Limit | Notes | +| ---------- | ------------ | ----------- | ------------------------------------------------------ | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -| 계층 | 일일 한도 | 비율 제한 | 메모 | -| ---- | ------------- | ---------------- | ---------------------------- | -| 무료 |**14,400RPD**| 모델당 30RPM | 신용카드 없음; 429 한도, 청구되지 않음 | +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -무료로 제공됨: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -| 모델 | 접두사 | 일일 무료 할당량 | 메모 | -| ---------------- | ------ | ----------------- | ---------- | -| 'LongCat-Flash-Lite' | `lc/` |**5천만 토큰**품목 | 사상 최대 규모의 무료 할당량 | -| 'LongCat-Flash-Chat' | `lc/` | 50만 토큰 | 다단계 채팅 | -| 'LongCat-플래시 사고' | `lc/` | 50만 토큰 | 추론 / CoT | -| 'LongCat-Flash-Thinking-2601' | `lc/` | 50만 토큰 | 2026년 1월 버전 | -| 'LongCat-Flash-Omni-2603' | `lc/` | 50만 토큰 | 다중 모드 | +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -> 공개 베타 버전에서는 100% 무료입니다. 이메일이나 전화로 [longcat.chat](https://longcat.chat)에 가입하세요. 매일 00:00 UTC에 재설정됩니다.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -| 모델 | 접두사 | 비율 제한 | 뒤에 공급자 | +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ------------- | ---------------- | ----------------------------------------- | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | + +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` + +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 + +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | + +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. + +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| '오픈아이' | `폴/` | 요청 1개/15초 | GPT-5 | -| '클로드' | `폴/` | 요청 1개/15초 | 인류학 클로드 | -| `쌍둥이자리` | `폴/` | 요청 1개/15초 | 구글 제미니 | -| '깊은 탐색' | `폴/` | 요청 1개/15초 | DeepSeek V3 | -| `라마` | `폴/` | 요청 1개/15초 | 메타 라마 4 스카우트 | -| '미스트랄' | `폴/` | 요청 1개/15초 | 미스트랄 AI | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**마찰 없음:**가입이나 API 키가 없습니다. 빈 키 필드가 있는 Pollinations 공급자를 추가하면 즉시 작동합니다.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| 계층 | 일일 뉴런 | 동등한 사용법 | 메모 | -| ---- | ------------- | -------------------------- | ---------- | -| 무료 |**10,000**| ~150 LLM 응답 / 500초 오디오 / 15K 임베드 | 글로벌 에지, 50개 이상의 모델 | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 -인기 있는 무료 모델: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo`(무료 오디오!), `@cf/qwen/qwen2.5-coder-15b-instruct` +| Tier | Daily Neurons | Equivalent Usage | Notes | +| ---- | ------------- | --------------------------------------- | ----------------------- | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -> [dash.cloudflare.com](https://dash.cloudflare.com)의 API 토큰 + 계정 ID가 필요합니다. 공급자 설정에 계정 ID를 저장합니다.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -| 계층 | 무료 할당량 | 위치 | 메모 | -| ---- | ------------- | ------------ | ---------------------- | -| 무료 |**100만 개의 토큰**| 🇫🇷 파리, EU | 한도 내에서는 신용카드가 필요하지 않습니다 | +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -무료로 이용 가능: `qwen3-235b-a22b-instruct-2507`(Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 -> EU/GDPR을 준수합니다. [console.scaleway.com](https://console.scaleway.com)에서 API 키를 받으세요. +| Tier | Free Quota | Location | Notes | +| ---- | ------------- | ------------ | ----------------------------------- | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | ->**💡 최고의 무료 스택(11개 제공자, 영원히 $0):** +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` + +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). + +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > ->```` -> 키로(kr/) → Claude Sonnet/Haiku UNLIMITED -> Qoder(if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -> LongCat Lite (lc/) → LongCat-Flash-Lite — 5천만 토큰/일 🔥 -> 수분(pol/) → GPT-5, Claude, DeepSeek, Llama 4 - 열쇠 필요 없음 -> Qwen(qw/) → qwen3-coder 모델 무제한 -> Gemini(gemini/) → Gemini 2.5 Flash — 1,500 요청/일 무료 -> Cloudflare AI(cf/) → 50개 이상의 모델 — 10,000개의 뉴런/일 -> Scaleway(scw/) → Qwen3 235B, Llama 70B — 1M 무료 토큰(EU) -> Groq(groq/) → Llama/Gemma — 14.4K 요청/일 초고속 -> NVIDIA NIM(nvidia/) → 70개 이상의 공개 모델 — 영원히 40RPM -> 대뇌(cerebras/) → Llama/Qwen 세계 최고 속도 — 100만 tok/일 ->````## 🎙️ Free Transcription Combo +> ``` +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` ->**$0**에 모든 오디오/비디오 전사 — Deepgram이 무료로 200달러, AssemblyAI가 50달러로 대체, Groq Whisper를 무제한 긴급 백업으로 제공합니다. +## 🎙️ Free Transcription Combo -| 공급자 | 무료 크레딧 | 최고의 모델 | 비율 제한 | -| ----------------- | --------- | ------------------------------- | --------------- | -| 🟢**딥그램**|**$200 무료**(가입) | `nova-3` — 최고의 정확도, 30개 이상의 언어 | 무료 크레딧에는 RPM 제한이 없습니다 | -| 🔵**AssemblyAI**|**$50 무료**(가입) | `universal-3-pro` — 챕터, 감정, PII | 무료 크레딧에는 RPM 제한이 없습니다 | -| 🔴**그로크**|**영원히 무료**| `whisper-large-v3` — OpenAI 속삭임 | 30RPM(속도 제한) | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. -**`/dashboard/combos`에 제안된 콤보:**``` +| Provider | Free Credits | Best Model | Rate Limit | +| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | + +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -````` +``` -그런 다음 `/dashboard/media` →**녹화**탭에서 오디오 또는 비디오 파일을 업로드하고 → 콤보 엔드포인트를 선택하고 → 지원되는 형식으로 변환을 가져옵니다.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0은 단순한 릴레이 프록시가 아닌 운영 플랫폼으로 구축되었습니다.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| 기능 | 그것이 하는 일 | -| ------------------------------- | ---------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Grok-4 빠른 제품군** | $0.20/$0.50/M의 xAI 모델 — 벤치마크 1143ms(Gemini 2.5 Flash보다 30% 빠름) | -| 🧠**Z.AI를 통한 GLM-5** | 128K 출력 컨텍스트, $0.5/1M — GLM 제품군의 최신 플래그십 | -| 🔮**미니맥스 M2.5** | $0.30/1M의 추론 + 에이전트 작업 — M2.1에서 대폭 업그레이드 | -| 🎯**모델별 toolCalling 플래그** | 레지스트리의 모델별 `toolCalling: true/false` — AutoCombo는 도구를 사용할 수 없는 모델을 건너뜁니다. | -| 🌍**다국어 의도 감지** | AutoCombo 채점의 PT/ZH/ES/AR 키워드 — 영어가 아닌 콘텐츠에 대한 더 나은 모델 선택 | -| 📊**벤치마크 기반 폴백** | 라이브 요청의 실제 p95 대기 시간은 콤보 점수를 제공합니다. AutoCombo는 실제 데이터에서 학습합니다 | -| 🔁**중복 제거 요청** | 콘텐츠 해시 기반 중복 제거 창 — 다중 에이전트 안전, 중복 청구 방지 | -| 🔌**플러그형 라우터 전략** | 확장 가능한 `RouterStrategy` 인터페이스 — 사용자 정의 라우팅 로직을 플러그인으로 추가 | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| 기능 | 그것이 하는 일 | -| -------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------- | -| 🎮**모델 놀이터** | 모든 모델을 직접 테스트할 수 있는 대시보드 페이지 — 제공자/모델/엔드포인트 선택기, 모나코 편집기, 스트리밍, 중단, 타이밍 | -| 🔏**CLI 지문 매칭** | 기본 CLI 서명과 일치하도록 공급자별 헤더/본문 순서 - 설정 > 보안에서 공급자별로 전환합니다.**프록시 IP는 보존됩니다** | -| 🤝**ACP 지원(에이전트 클라이언트 프로토콜)** | CLI 에이전트 검색(Codex, Claude, Goose, Gemini CLI, OpenClaw + 9개 이상), 프로세스 생성기, `/api/acp/agents` 엔드포인트 | -| 🤖**ACP 상담원 대시보드** | 디버그 › 에이전트 페이지 — 모든 CLI 도구에 대한 설치 상태, 버전, 사용자 지정 에이전트 양식이 포함된 14개 에이전트의 그리드입니다.**OpenCode**사용자에게는 사용 가능한 모든 모델과 함께 즉시 사용 가능한 구성을 자동 생성하는 "opencode.json 다운로드" 버튼이 표시됩니다. | -| 🔧**사용자 정의 모델 `apiFormat` 라우팅** | `apiFormat: "responses"`를 사용하는 사용자 정의 모델은 이제 Responses API 변환기 | -| 🏢**Codex 작업 공간 격리** | 이메일당 여러 Codex 작업 공간 — OAuth는 작업 공간 ID로 연결을 올바르게 구분합니다 | -| 🔄**전자 자동 업데이트** | 데스크탑 앱에서 업데이트 확인 + 재시작 시 자동 설치 | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| 기능 | 그것이 하는 일 | -| --------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**MCP 서버(25개 도구)** | 3가지 전송을 통한 IDE/에이전트 도구: stdio, SSE(`/api/mcp/sse`), 스트리밍 가능한 HTTP(`/api/mcp/stream`). 18개 코어 + 3개 메모리 + 4개 스킬 툴 | -| 🤝**A2A 서버(JSON-RPC + SSE)** | 동기화 및 스트리밍 흐름을 통한 에이전트 간 작업 실행 | -| 🧭**통합 엔드포인트 페이지** | 엔드포인트 프록시, MCP, A2A 및 API 엔드포인트 탭이 있는 탭 관리 페이지 | -| 🎚️**서비스 활성화/비활성화 토글** | 설정 지속성을 갖춘 MCP 및 A2A용 ON/OFF 스위치(기본값: OFF) | -| 🛰️**MCP 런타임 하트비트** | 실제 프로세스 상태(pid, 가동 시간, 하트비트 기간, 전송, 범위 모드) | -| 📋**MCP 감사 추적** | 성공/실패 및 주요 속성이 포함된 필터링 가능한 감사 로그 | -| 🔐**MCP 범위 적용** | 제어된 도구 액세스를 위한 10개의 세부적인 범위 권한 | -| 📡**A2A 작업 수명주기 관리** | 작업 나열/필터링, 이벤트/아티팩트 검사, 실행 중인 작업 취소 | -| 📋**에이전트 카드 검색** | 클라이언트 자동 검색을 위한 `/.well-known/agent.json` | -| 🧪**프로토콜 E2E 테스트 하네스** | 실제 MCP SDK + A2A 클라이언트는 `test:protocols:e2e`로 흐릅니다 | -| ⚙️**작동 제어** | 하나의 제어 표면에서 콤보 전환, 탄력성 프로필 적용, 차단기 재설정 | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| 기능 | 그것이 하는 일 | -| ----------------------------- | ----------------------------------------------------------------- | ----------------------- | -| 🎯**스마트 4계층 폴백** | 자동 경로: 구독 → API 키 → 저렴한 → 무료 | -| 📊**실시간 할당량 추적** | 라이브 토큰 수 + 공급자별 카운트다운 재설정 | -| 🔄**형식 번역** | OpenAI ⇔ Claude ⇔ Gemini ⇔ 스키마 안전 변환을 통한 응답 | -| 👥**다중 계정 지원** | 지능적인 선택을 통해 공급자당 여러 계정 | -| 🔄**자동 토큰 새로고침** | 재시도 시 OAuth 토큰이 자동으로 새로 고쳐집니다. | -| 🎨**맞춤형 콤보** | 9가지 밸런싱 전략 + 폴백 체인 제어 | -| 🌐**와일드카드 라우터** | `provider/*` 동적 라우팅 | -| 🧠**사고 예산 통제** | 통과, 자동, 사용자 정의 및 적응형 추론 제한 | -| 🔀**모델 별칭** | 내장 + 사용자 정의 모델 앨리어싱 및 마이그레이션 안전 | -| ⚡**배경 저하** | 우선순위가 낮은 백그라운드 작업을 저렴한 모델로 라우팅 | -| 🧪**작업 인식 스마트 라우팅** | 콘텐츠 유형별 모델 자동 선택(코딩/비전/분석/요약) | -| 🔄**A2A 에이전트 워크플로** | 상태 저장 다단계 에이전트 실행을 위한 결정적 FSM 조정자 | -| 🔀**적응형 라우팅** | 토큰 볼륨 및 프롬프트 복잡성을 기반으로 한 동적 전략 재정의 | -| 🎲**공급자 다양성** | Shannon 엔트로피 점수 균형 자동 콤보 트래픽 분배 | -| 💬**시스템 프롬프트 삽입** | 일관되게 적용되는 글로벌 행동 제어 | -| 📄**응답 API 호환성** | Codex 및 고급 에이전트 작업 흐름에 대한 전체 `/v1/responses` 지원 | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| 기능 | 그것이 하는 일 | -| ---------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**이미지 생성** | 클라우드 및 로컬 백엔드가 포함된 `/v1/images/세대` | -| 📐**임베딩** | 검색 및 RAG 파이프라인용 `/v1/embeddings` | -| 🎤**오디오 전사** | `/v1/audio/transcriptions` — 7개 공급자(Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), 자동 언어 감지, MP4/MP3/WAV 지원 | -| 🔊**텍스트 음성 변환** | `/v1/audio/speech` — 올바른 오류 메시지가 있는 10개 공급자(ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) | -| 🎬**비디오 생성** | `/v1/videos/세대`(ComfyUI + SD WebUI 워크플로) | -| 🎵**음악 세대** | `/v1/music/세대`(ComfyUI 워크플로) | -| 🛡️**조정** | `/v1/moderations` 안전 확인 | -| 🔀**재순위** | 관련성 점수를 위한 `/v1/rerank` | -| 🔍**웹 검색**🆕 | `/v1/search` — 5개 공급자(Serper, Brave, Perplexity, Exa, Tavily), 월 6,500개 이상의 무료, 자동 장애 조치, 캐시 | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| 기능 | 그것이 하는 일 | -| --------------------------------- | -------------------------------------------------------------------------- | -------------------------------- | -| 🔌**회로 차단기** | 임계값 제어를 통한 모델별 트립/복구 | -| 🎯**엔드포인트 인식 모델** | 사용자 정의 모델은 지원되는 엔드포인트 + API 형식을 선언합니다 | -| 🛡️**천둥 방지 무리** | 재시도/비율 이벤트에 대한 뮤텍스 + 세마포어 보호 | -| 🧠**의미 + 서명 캐시** | 두 개의 캐시 레이어로 비용/지연 시간 감소 | -| ⚡**멱등성 요청** | 중복 보호 창 | -| 🔒**TLS 지문 스푸핑** | 브라우저와 유사한 TLS 지문 —**봇 감지 및 계정 플래그 지정 감소** | -| 🔏**CLI 지문 매칭** | 기본 CLI 요청 서명과 일치 —**프록시 IP를 보존하면서 금지 위험을 줄입니다** | -| 🌐**IP 필터링** | 노출된 배포에 대한 허용 목록/차단 목록 제어 | -| 📊**편집 가능한 속도 제한** | 지속성을 통해 구성 가능한 전역/공급자 수준 제한 | -| 📉**우아한 저하** | 핵심 게이트웨이 작업을 보호하는 다계층 기능 대체 | -| 📜**구성 감사 추적** | 간단한 롤백으로 운영 드리프트를 방지하는 Diff 기반 변경 추적 | -| ⏳**공급자 상태 동기화** | 승인 실패 전에 경고를 트리거하는 사전 예방적 토큰 만료 모니터링 | -| 🚪**금지된 계정 자동 비활성화** | 영구 차단된 토큰 계정을 자동으로 봉인하는 작동 회로 차단기 | -| 🔑**API 키 관리 + 범위 지정** | 보안 키 발급/교체 및 모델/공급자 제어 | -| 👁️**범위가 지정된 API 키 공개**🆕 | `ALLOW_API_KEY_REVEAL`을 통한 API 키 복구 선택 | -| 🛡️**보호된 `/models`** | 모델 카탈로그에 대한 선택적 인증 게이팅 및 공급자 숨기기 | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| 기능 | 그것이 하는 일 | -| ---------------------------- | ------------------------------------------------- | ---------------------------- | -| 📝**요청 + 프록시 로깅** | 전체 요청/응답 및 프록시 로깅 | -| 📉**스트리밍된 상세 로그**🆕 | SSE 페이로드 스트림을 UI로 깔끔하게 재구성 | -| 📋**통합 로그 대시보드** | 한 페이지에서 요청, 프록시, 감사 및 콘솔 보기 | -| 🔍**원격 측정 요청** | p50/p95/p99 대기 시간 및 요청 추적 | -| 🏥**건강 대시보드** | 가동 시간, 차단기 상태, 잠금, 캐시 통계 | -| 💰**비용 추적** | 예산 제어 및 모델별 가격 가시성 | -| 📈**분석 시각화** | 모델/공급자 사용량 통찰력 및 추세 보기 | -| 🧪**평가 프레임워크** | 구성 가능한 매치 전략을 사용한 골든 세트 테스트 | -| 📡**실시간 진단**🆕 | 정확한 콤보 라이브 테스트를 위한 시맨틱 캐시 우회 | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| 기능 | 그것이 하는 일 | -| --------------------------------- | ---------------------------------------------------------------- | --------------------- | -| 🌐**어디서나 배포** | 로컬호스트, VPS, Docker, 클라우드 환경 | -| 🚇**Cloudflare 터널**🆕 | 대시보드에서 원클릭 Quick Tunnel 통합 | -| 🔑**API 키 모델 필터링** | 할당된 Bearer 컨텍스트 역할을 통해 필터링된 기본 /v1/models 응답 | -| ⚡**스마트 캐시 우회** | 구성 가능한 TTL 휴리스틱 및 강제 재페치 제어 | -| 🔄**백업/복원** | 내보내기/가져오기 및 재해 복구 흐름 | -| 🧙**온보딩 마법사** | 첫 실행 안내 설정 | -| 🔧**CLI 도구 대시보드** | 널리 사용되는 코딩 도구를 위한 원클릭 설정 | -| 🎮**모델 놀이터** | 대시보드에서 공급자/모델/엔드포인트 테스트 | -| 🔏**CLI 지문 토글** | 설정 > 보안 | 공급자별 지문 일치 | -| 🌐**i18n(30개 언어)** | RTL 적용 범위를 포함한 전체 대시보드 + 문서 언어 지원 | -| 🧹**모든 모델 지우기** | 공급자 세부정보에서 원클릭 모델 목록 삭제 | -| 👁️**사이드바 컨트롤**🆕 | 모양 설정에서 구성요소 및 통합 숨기기 | -| 📋**이슈 템플릿** | 버그 및 기능을 위한 표준화된 GitHub 템플릿 | -| 📂**사용자 정의 데이터 디렉터리** | 저장 위치에 대한 `DATA_DIR` 재정의 | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1298,105 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -할당량, 속도 또는 상태가 실패하면 OmniRoute는 수동 전환 없이 자동으로 다음 후보로 이동합니다.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A는 UI 및 문서에서 검색 가능합니다(숨겨지지 않음). -- 프로토콜 상태 API는 실시간 운영 데이터(`/api/mcp/*`, `/api/a2a/*`)를 노출합니다. -- 대시보드에는 2일 차 작업에 대한 작업이 포함됩니다(콤보 토글, 차단기 재설정, 작업 취소).#### Translator + validation workflow +#### Protocol management that is visible and operable -번역기 영역에는 다음이 포함됩니다. +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**플레이그라운드**: 변환 확인 요청 -**채팅 테스터**: 전체 요청/응답 왕복 -**테스트 벤치**: 한 번에 여러 사례 실행 -**라이브 모니터**: 실시간 교통 상황 보기 +#### Translator + validation workflow -또한 `npm run test:protocols:e2e`를 통해 실제 클라이언트를 사용한 프로토콜 검증도 가능합니다. +The Translator area includes: -> 📖**[MCP 서버 README](open-sse/mcp-server/README.md)**— 도구 참조, IDE 구성 및 클라이언트 예제 +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A 서버 README](src/lib/a2a/README.md)**— 기술, JSON-RPC 방법, 스트리밍 및 작업 수명 주기## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute에는 골든 세트에 대해 LLM 응답 품질을 테스트하기 위한 내장 평가 프레임워크가 포함되어 있습니다. 대시보드의**분석 → 평가**를 통해 액세스하세요.### Built-in Golden Set +## 🧪 Evaluations (Evals) -사전 로드된 "OmniRoute Golden Set"에는 다음에 대한 테스트 사례가 포함되어 있습니다. +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- 인사말, 수학, 지리, 코드 생성 -- JSON 형식 준수, 번역, 마크다운 생성 -- 안전 거부(유해 콘텐츠), 카운팅, 부울 논리### Evaluation Strategies +### Built-in Golden Set -| 전략 | 설명 | 예 | -| ---------- | -------------------------------------------------------------- | -------------------------- | --- | -| '정확하다' | 출력은 정확히 일치해야 합니다 | ``4"` | -| `포함` | 출력에는 하위 문자열(대소문자 구분 안 함)이 포함되어야 합니다. | ``파리"` | -| `정규식` | 출력은 정규식 패턴과 일치해야 합니다 | ``1.*2.*3"` | -| '맞춤형' | 사용자 정의 JS 함수가 true/false를 반환합니다. | `(출력) => 출력.길이 > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) -<상세> +
+🧩 MCP Setup (Model Context Protocol) -🧩 MCP 설정(모델 컨텍스트 프로토콜) +Start MCP transport in stdio mode: -stdio 모드에서 MCP 전송을 시작합니다.```bash +```bash omniroute --mcp +``` -```` +Recommended validation flow: -권장되는 검증 흐름: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. stdio를 통해 MCP 클라이언트를 연결합니다. -2. `omniroute_get_health`를 실행합니다. -3. `omniroute_list_combos`를 실행합니다. -4. `/dashboard/mcp`를 열어 하트비트, 활동 및 감사를 확인합니다. - -자동화에 유용한 API: +Useful APIs for automation: - `GET /api/mcp/status` - `GET /api/mcp/tools` - `GET /api/mcp/audit` -- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats` -<상세> -🤝 A2A 설정(Agent2Agent) + -에이전트를 검색합니다.```bash +
+🤝 A2A Setup (Agent2Agent) + +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -작업 보내기:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` - -수명주기 관리: +Manage lifecycle: - `GET /api/a2a/status` - `GET /api/a2a/tasks` - `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -운영 UI: +Operational UI: -- 작업/상태/스트림 관찰 가능성 및 연기 작업을 위한 `/dashboard/a2a`
+- `/dashboard/a2a` for task/state/stream observability and smoke actions -<상세> -🧪 엔드투엔드 프로토콜 검증 + -실제 클라이언트에서 두 프로토콜을 모두 검증합니다.```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -이는 다음을 확인합니다. +This verifies: -- MCP SDK 클라이언트 연결/목록/호출 -- A2A 검색/전송/스트리밍/가져오기/취소 -- MCP 감사 및 A2A 작업 관리 API의 데이터 교차 확인
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs -<상세> + -💳 구독 제공업체### Claude Code (Pro/Max) +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1409,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**프로 팁:**복잡한 작업에는 Opus를 사용하고, 속도를 높이려면 Sonnet을 사용하세요. OmniRoute는 모델당 할당량을 추적합니다!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1423,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -이제 각 Codex 계정에는 `대시보드 -> 공급자`에 정책 토글이 있습니다. +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h`(ON/OFF): 5시간 창 임계값 정책을 시행합니다. -- '주간'(ON/OFF): 주간 창 임계값 정책을 시행합니다. -- 임계값 동작: 활성화된 창이 사용량의 90% 이상에 도달하면 해당 계정을 건너뜁니다. -- 순환 동작: OmniRoute는 자동으로 다음 적격 Codex 계정으로 라우팅합니다. -- 재설정 동작: 공급자 'resetAt' 시간이 지나면 계정이 자동으로 다시 자격을 갖추게 됩니다. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -시나리오: +Scenarios: -- `5h ON` + `Weekly ON`: 두 창이 임계값에 도달하면 계정을 건너뜁니다. -- `5h OFF` + `Weekly ON`: 주간 사용량만 계정을 차단할 수 있습니다. -- `5h ON` + `Weekly OFF`: 5시간 동안만 계정을 차단할 수 있습니다. -- `resetAt` 통과: 계정이 자동으로 순환에 다시 들어갑니다(수동 재활성화 없음).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1448,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**최고의 가치:**엄청난 무료 등급! 유료 등급 이전에 사용하세요.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1463,74 +1662,91 @@ Models:
-<상세> +
+🔑 API Key Providers -🔑 API 키 제공자### NVIDIA NIM (FREE developer access — 70+ models) +### NVIDIA NIM (FREE developer access — 70+ models) -1. 가입: [build.nvidia.com](https://build.nvidia.com) -2. 무료 API 키 받기(1000 추론 크레딧 포함) -3. 대시보드 → 공급자 추가 → NVIDIA NIM: - - API 키: `nvapi-your-key` +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**모델:**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` 외 50개 이상 +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -**프로 팁:**OpenAI 호환 API — OmniRoute의 형식 변환과 원활하게 작동합니다!### DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -1. 회원가입: [platform.deepseek.com](https://platform.deepseek.com) -2. API 키 받기 -3. 대시보드 → 공급자 추가 → DeepSeek +### DeepSeek -**모델:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -1. 회원가입: [console.groq.com](https://console.groq.com) -2. API 키 받기(무료 등급 포함) -3. 대시보드 → 공급자 추가 → Groq +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**모델:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +### Groq (Free Tier Available!) -**프로 팁:**초고속 추론 — 실시간 코딩에 가장 적합합니다!### OpenRouter (100+ Models) +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -1. 회원가입: [openrouter.ai](https://openrouter.ai) -2. API 키 받기 -3. 대시보드 → 공급자 추가 → OpenRouter +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**모델:**단일 API 키를 통해 모든 주요 제공업체의 100개 이상의 모델에 액세스할 수 있습니다. +**Pro Tip:** Ultra-fast inference — best for real-time coding! -**대시보드 동작:**OpenRouter 모델은**사용 가능한 모델**에서 관리됩니다. 수동 추가, 가져오기 및 자동 동기화는 모두 동일한 목록을 업데이트합니다.
+### OpenRouter (100+ Models) -<상세> +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -💰 저렴한 제공업체(백업)### GLM-4.7 (Daily reset, $0.6/1M) +**Models:** Access 100+ models from all major providers through a single API key. -1. 회원가입: [Zhipu AI](https://open.bigmodel.cn/) -2. Coding Plan에서 API Key 받기 -3. 대시보드 → API 키 추가: - - 제공자: `glm` - - API 키: `your-key` +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -**사용:**`glm/glm-4.7` + -**프로 팁:**코딩 계획은 1/7 비용으로 3배 할당량을 제공합니다! 매일 오전 10시에 초기화됩니다.### MiniMax M2.1 (5h reset, $0.20/1M) +
+💰 Cheap Providers (Backup) -1. 회원가입: [미니맥스](https://www.minimax.io/) -2. API 키 받기 -3. 대시보드 → API Key 추가 +### GLM-4.7 (Daily reset, $0.6/1M) -**사용:**`minimax/MiniMax-M2.1` +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**프로 팁:**긴 컨텍스트(100만 토큰)에 대한 가장 저렴한 옵션!### Kimi K2 ($9/month flat) +**Use:** `glm/glm-4.7` -1. 구독: [문샷 AI](https://platform.moonshot.ai/) -2. API 키 받기 -3. 대시보드 → API Key 추가 +**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. -**용도:**`kimi/kimi-latest` +### MiniMax M2.1 (5h reset, $0.20/1M) -**프로 팁:**1,000만 토큰에 대해 월 $9 고정 = 유효 비용 $0.90/1M!
+1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key -<상세> +**Use:** `minimax/MiniMax-M2.1` -🆓 무료 제공업체(긴급 백업)### Qoder (5 FREE models via OAuth) +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1571,9 +1787,10 @@ Models:
-<상세> +
+🎨 Create Combos -🎨 콤보 만들기### Example 1: Maximize Subscription → Cheap Backup +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1601,9 +1818,10 @@ Cost: $0 forever!
-<상세> +
+🔧 CLI Integration -🔧 CLI 통합### Cursor IDE +### Cursor IDE ``` Settings → Models → Advanced: @@ -1614,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -원클릭 구성을 위해 대시보드의**CLI 도구**페이지를 사용하거나 `~/.claude/settings.json`을 수동으로 편집하세요.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1625,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**옵션 1 - 대시보드(권장):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**옵션 2 — 수동:**`~/.openclaw/openclaw.json`을 편집합니다.```json +```json { "models": { "providers": { @@ -1642,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **참고:**OpenClaw는 로컬 OmniRoute에서만 작동합니다. IPv6 해결 문제를 방지하려면 `localhost` 대신 `127.0.0.1`을 사용하세요.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1656,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**1단계:**OmniRoute를 사용자 지정 공급자로 추가합니다.```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**2단계:**프로젝트 루트에서 `opencode.json`을 생성/편집합니다.```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1682,118 +1909,130 @@ opencode } } } -```` +``` -**3단계:**OpenCode에서 모델을 선택합니다.```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**팁:**OmniRoute `/v1/models` 엔드포인트에서 사용 가능한 모델을 `models` 섹션에 추가하세요. OmniRoute 대시보드에서 'provider/model-id' 형식을 사용하세요.
+ --- ## 문제 해결 -<상세> -문제해결 가이드를 펼치려면 클릭하세요. +
+Click to expand troubleshooting guide -**"언어 모델이 메시지를 제공하지 않았습니다"** +**"Language model did not provide messages"** -- 공급자 할당량 소진 → 대시보드 할당량 추적기 확인 -- 해결 방법: 콤보 폴백을 사용하거나 더 저렴한 계층으로 전환하세요. +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**비율 제한** +**Rate limiting** -- 구독 할당량 초과 → GLM/MiniMax로 대체 -- 콤보 추가: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**OAuth 토큰이 만료되었습니다** +**OAuth token expired** -- OmniRoute에 의해 자동 새로고침 -- 문제가 지속되는 경우: Dashboard → Provider → Reconnect +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**높은 비용** +**High costs** -- 대시보드 → 비용에서 사용 통계를 확인하세요. -- 기본 모델을 GLM/MiniMax로 전환 -- 중요하지 않은 작업에는 무료 계층(Gemini CLI, Qoder)을 사용합니다. +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**대시보드/API 포트가 잘못되었습니다** +**Dashboard/API ports are wrong** -- `PORT`는 표준 기본 포트(및 기본적으로 API 포트)입니다. -- `API_PORT`는 OpenAI 호환 API 리스너만 재정의합니다. -- `DASHBOARD_PORT`는 대시보드/Next.js 리스너만 재정의합니다. -- 'NEXT_PUBLIC_BASE_URL'을 대시보드/공개 URL로 설정합니다(OAuth 콜백용). +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**클라우드 동기화 오류** +**Cloud sync errors** -- 'BASE_URL'이 실행 중인 인스턴스를 가리키는지 확인하세요. -- 'CLOUD_URL'이 예상 클라우드 엔드포인트를 가리키는지 확인하세요. -- `NEXT_PUBLIC_*` 값을 서버 측 값에 맞게 유지합니다. +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**첫 번째 로그인이 작동하지 않습니다** +**First login not working** -- `.env`에서 `INITIAL_PASSWORD`를 확인하세요. -- 설정되지 않은 경우 대체 비밀번호는 '123456'입니다. +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**요청 로그 없음** +**No request logs** -- 요청 아티팩트는 요청당 하나의 JSON 파일로 `DATA_DIR/call_logs/`에 기록됩니다. -- 자세한 단계별 페이로드가 필요한 경우 대시보드 → 로그 → 요청 로그에서 파이프라인 캡처를 활성화합니다. -- `logs/application/app.log`에 애플리케이션 콘솔 로그도 저장하려면 `APP_LOG_TO_FILE=true`를 설정하세요. -- 필요에 따라 `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` 및 `CALL_LOG_MAX_ENTRIES`를 조정합니다. +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**OpenAI 호환 공급자에 대해 연결 테스트에서 "잘못됨"이 표시됨** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- 많은 공급자가 '/models' 엔드포인트를 노출하지 않습니다. -- OmniRoute v1.0.6+에는 채팅 완료를 통한 대체 검증이 포함되어 있습니다. -- 기본 URL에 '/v1' 접미사가 포함되어 있는지 확인하세요.### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ VPS, Docker 또는 원격 서버에서 OmniRoute를 실행하는 사용자에게 중요**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -**Antigravity**및**Gemini CLI**공급자는**Google OAuth 2.0**을 사용합니다. Google에서는 앱의 Google Cloud Console에 사전 등록된 URI 중 하나와 정확히 일치하도록 OAuth 흐름의 'redirect_uri'를 요구합니다. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -OmniRoute에 번들로 제공되는 OAuth 자격 증명은**`localhost`에만 등록됩니다**. 원격 서버(예: `https://omniroute.myserver.com`)에서 OmniRoute에 액세스하면 Google은 다음을 통한 인증을 거부합니다.``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -서버의 URI를 사용하여 Google Cloud Console에서**OAuth 2.0 클라이언트 ID**를 만들어야 합니다.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Google Cloud Console 열기** +#### Step-by-step -이동: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. 새 OAuth 2.0 클라이언트 ID 생성** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) --**"+ 자격 증명 만들기"**→**"OAuth 클라이언트 ID"**를 클릭합니다. +**2. Create a new OAuth 2.0 Client ID** -- 애플리케이션 유형:**"웹 애플리케이션"** -- 이름: 원하는 것(예: `OmniRoute Remote`) +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -**3. 승인된 리디렉션 URI 추가** +**3. Add Authorized Redirect URIs** -**"승인된 리디렉션 URI"**필드에 다음을 추가합니다.``` +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> `your-server.com`을 서버의 도메인 또는 IP로 바꿉니다(필요한 경우 포트 포함, 예: `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. 자격 증명 저장 및 복사** +After creating, Google will show the **Client ID** and **Client Secret**. -생성 후 Google은**클라이언트 ID**및**클라이언트 비밀번호**를 표시합니다. +**5. Set environment variables** -**5. 환경 변수 설정** +In your `.env` (or Docker environment variables): -`.env`(또는 Docker 환경 변수)에서:```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1802,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. OmniRoute 다시 시작**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. 다시 연결해 보세요** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -대시보드 → 공급자 → Antigravity(또는 Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -이제 Google은 `https://your-server.com/callback`으로 올바르게 리디렉션됩니다.--- +--- #### Temporary workaround (without custom credentials) -지금 바로 자격 증명을 설정하고 싶지 않은 경우에도**수동 URL 흐름**을 사용할 수 있습니다. +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute는 Google 인증 URL을 엽니다. -2. 승인 후 Google은 `localhost`로 리디렉션을 시도합니다(원격 서버에서는 실패함). -3. 브라우저의 주소 표시줄에서**전체 URL을 복사**하세요(페이지가 로드되지 않는 경우에도 해당). -4. 해당 URL을 OmniRoute 연결 모달에 표시된 필드에 붙여넣습니다. -5.**"연결"**을 클릭하세요. +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> 이는 리디렉션 페이지 로드 여부에 관계없이 URL의 인증 코드가 유효하기 때문에 작동합니다.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. -<상세> -🇧🇷 포르투갈어 버전#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Os는**반중력**및**Gemini CLI**를 인증하기 위해**Google OAuth 2.0**을 입증했습니다. O Google은 'redirect_uri'를 사용하여 Fluxo를 사용하지 않습니다. OAuth seja**시험**uma das URIs pre-cadastradas no Google Cloud Console do aplicativo. +
+🇧🇷 Versão em Português -OAuth는 OmniRoute에 대한 신용 정보를 포함하지 않으므로**`localhost`**에 대한 액세스 권한을 갖습니다. 액세스 권한이 있는 OmniRoute em um servidor remoto(예: `https://omniroute.meuservidor.com`), Google rejeita a autenticação com:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Você precisa criar um**OAuth 2.0 Client ID**no Google Cloud Console com a URI do seu servidor.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Google Cloud Console에 액세스** +#### Passo a passo -아브라: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Acesse o Google Cloud Console** -**2. Crie um novo OAuth 2.0 클라이언트 ID** +Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- em 클릭**"+ 자격 증명 만들기"**→**"OAuth 클라이언트 ID"** -- 적용 분야:**"웹 애플리케이션"** -- 이름: escolha qualquer nome (예: `OmniRoute Remote`) +**2. Crie um novo OAuth 2.0 Client ID** -**3. 승인된 리디렉션 URI로서의 Adicione** +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -아니요**"승인된 리디렉션 URI"**, 추가:``` +**3. Adicione as Authorized Redirect URIs** + +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> `seu-servidor.com` pelo domínio 또는 IP do seu servidor로 대체합니다(필요한 포트 포함, 예: `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. 사본을 자격 증명으로 저장** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Após criar, o Google morerá o**클라이언트 ID**e o**클라이언트 비밀번호**. +**5. Configure as variáveis de ambiente** -**5. 다양한 주변 환경으로 구성** +No seu `.env` (ou nas variáveis de ambiente do Docker): -`.env`가 없습니다(Docker의 다양한 주변 환경).```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1881,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reinicie 또는 OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute - -```` +``` **7. Tente conectar novamente** -대시보드 → 공급자 → Antigravity(또는 Gemini CLI) → OAuth +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Agora에서는 'https://seu-servidor.com/callback' 및 자동 기능으로 Google을 리디렉션합니다.--- +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. + +--- #### Workaround temporário (sem configurar credenciais próprias) -Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo**manual de URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. OmniRoute는 Google 자동 생성 URL을 확인합니다. -2. Após você autorizar, o Google Tentará redirectionar para `localhost` (que falha no servidor remoto) -3.**URL 복사 완료**da barra de endereço do seu browser (mesmo que a página não carregue) +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) 4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute -5. Clique em**"연결"** +5. Clique em **"Connect"** -> 이 해결 방법은 URL을 통해 자동으로 코드를 확인하거나 독립적으로 리디렉션할 수 있도록 하는 것입니다.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1919,64 +2171,73 @@ Se não quiser criar credenciais próprias agora, ainda é possível usar o flux ## 🛠️ Tech Stack -<상세> -기술 스택 세부정보를 펼치려면 클릭하세요. +
+Click to expand tech stack details --**런타임**: Node.js 18–22 LTS(⚠️ Node.js 24+는**지원되지 않음**- `better-sqlite3` 네이티브 바이너리는 호환되지 않음) --**언어**: TypeScript 5.9 — `src/` 및 `open-sse/` 전체에서**100% TypeScript**(v2.0 이후 핵심 모듈에서는 `any`가 0임) --**프레임워크**: Next.js 16 + React 19 + Tailwind CSS 4 --**데이터베이스**: LowDB(JSON) + SQLite(도메인 상태 + 프록시 로그 + MCP 감사 + 라우팅 결정) --**스키마**: Zod(MCP 도구 I/O 검증, API 계약) --**프로토콜**: MCP(stdio/HTTP) + A2A v0.3(JSON-RPC 2.0 + SSE) --**스트리밍**: 서버에서 보낸 이벤트(SSE) --**인증**: OAuth 2.0(PKCE) + JWT + API 키 + MCP 범위 승인 --**테스트**: Node.js 테스트 실행기 + Vitest(단위, 통합, E2E를 포함한 900개 이상의 테스트) --**CI/CD**: GitHub Actions(자동 npm 게시 + 출시 시 Docker Hub) --**웹사이트**: [omniroute.online](https://omniroute.online) --**패키지**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**복원력**: 회로 차단기, 지수 백오프, 천둥 방지 무리, TLS 스푸핑, 자동 콤보 자가 치유
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## 문서 -| 문서 | 설명 | -| --------------------------------- | -------------------------------------- | -| [사용자 가이드](docs/USER_GUIDE.md) | 공급자, 콤보, CLI 통합, 배포 | -| [API 참조](docs/API_REFERENCE.md) | 예제가 포함된 모든 엔드포인트 | -| [MCP 서버](open-sse/mcp-server/README.md) | 16개의 MCP 도구, IDE 구성, Python/TS/Go 클라이언트 | -| [A2A 서버](src/lib/a2a/README.md) | JSON-RPC 2.0 프로토콜, 스킬, 스트리밍, 작업 관리 | -| [자동 콤보 엔진](docs/auto-combo.md) | 6단계 채점, 모드 팩, 자가 치유 | -| [문제 해결](docs/TROUBLESHOOTING.md) | 일반적인 문제 및 해결 방법 | -| [건축](docs/ARCHITECTURE.md) | 시스템 아키텍처 및 내부 | -| [기여](CONTRIBUTING.md) | 개발 설정 및 지침 | -| [OpenAPI 사양](docs/openapi.yaml) | OpenAPI 3.0 사양 | -| [보안정책](SECURITY.md) | 취약점 보고 및 보안 관행 | -| [VM 배포](docs/VM_DEPLOYMENT_GUIDE.md) | 전체 가이드: VM + nginx + Cloudflare 설정 | -| [기능 갤러리](docs/FEATURES.md) | 스크린샷을 포함한 시각적 대시보드 둘러보기 | -| [출시 체크리스트](docs/RELEASE_CHECKLIST.md) | 출시 전 유효성 검사 단계 |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute에는 여러 개발 단계에 걸쳐**210개 이상의 기능이 계획되어 있습니다**. 주요 영역은 다음과 같습니다. +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| 카테고리 | 계획된 기능 | 하이라이트 | -| ---------------- | ---------------- | ------------------------------------------------------------------------- | -| 🧠**라우팅 및 인텔리전스**| 25세 이상 | 최저 대기 시간 라우팅, 태그 기반 라우팅, 실행 전 할당량, P2C 계정 선택 | -| 🔒**보안 및 규정 준수**| 20세 이상 | SSRF 강화, 자격 증명 클로킹, 엔드포인트당 속도 제한, 관리 키 범위 지정 | -| 📊**관측성**| 15세 이상 | OpenTelemetry 통합, 실시간 할당량 모니터링, 모델별 비용 추적 | -| 🔄**공급자 통합**| 20세 이상 | 동적 모델 레지스트리, 공급자 쿨다운, 다중 계정 Codex, Copilot 할당량 구문 분석 | -| ⚡**성능**| 15세 이상 | 듀얼 캐시 레이어, 프롬프트 캐시, 응답 캐시, 스트리밍 Keepalive, 배치 API | -| 🌐**생태계**| 10세 이상 | WebSocket API, 구성 핫 리로드, 분산 구성 저장소, 상용 모드 |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**OpenCode 통합**— OpenCode AI 코딩 IDE에 대한 기본 공급자 지원 -- 🔗**TRAE 통합**— TRAE AI 개발 프레임워크를 완벽하게 지원 -- 📦**Batch API**— 대량 요청에 대한 비동기식 일괄 처리 -- 🎯**태그 기반 라우팅**— 사용자 정의 태그 및 메타데이터를 기반으로 요청 라우팅 -- 💰**최저 비용 전략**— 가장 저렴한 제공업체를 자동으로 선택합니다. +### 🔜 Coming Soon -> 📝 [`docs/new-features/`](docs/new-features/)에서 전체 기능 사양 확인 가능(217개 세부 사양)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1984,18 +2245,20 @@ OmniRoute에는 여러 개발 단계에 걸쳐**210개 이상의 기능이 계 ### How to Contribute -1. 저장소 포크 -2. 기능 브랜치를 생성합니다(`git checkout -b feature/amazing-feature`) -3. 변경 사항을 커밋합니다(`git commit -m 'Add amazing feature'`) -4. 브랜치로 푸시(`git push 원점 기능/amazing-feature`) -5. 풀 리퀘스트 열기 +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -자세한 지침은 [CONTRIBUTING.md](CONTRIBUTING.md)를 참조하세요.### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -2007,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -이 포크에 영감을 준 원본 프로젝트인**[decolua](https://github.com/decolua)**의**[9router](https://github.com/decolua/9router)**에게 특별히 감사드립니다. OmniRoute는 추가 기능, 다중 모드 API 및 전체 TypeScript 재작성을 통해 놀라운 기반을 구축합니다. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -이 JavaScript 포트에 영감을 준 최초의 Go 구현인**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**에게 특별히 감사드립니다.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## 라이선스 -MIT 라이선스 - 자세한 내용은 [LICENSE](LICENSE)를 참조하세요.--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/ko/docs/ARCHITECTURE.md b/docs/i18n/ko/docs/ARCHITECTURE.md index 309068038d..df58c9e9e4 100644 --- a/docs/i18n/ko/docs/ARCHITECTURE.md +++ b/docs/i18n/ko/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_최종 업데이트 날짜: 2026-03-28_## Executive Summary -OmniRoute는 Next.js를 기반으로 구축된 로컬 AI 라우팅 게이트웨이이자 대시보드입니다. -단일 OpenAI 호환 엔드포인트(`/v1/*`)를 제공하고 변환, 대체, 토큰 새로 고침 및 사용량 추적을 통해 여러 업스트림 공급자 간에 트래픽을 라우팅합니다. -핵심 기능: +_Last updated: 2026-03-28_ -- CLI/도구용 OpenAI 호환 API 표면(28개 공급자) -- 공급자 형식에 따른 요청/응답 번역 -- 모델 콤보 대체(다중 모델 시퀀스) -- 계정 수준 대체(제공업체당 다중 계정) -- OAuth + API 키 공급자 연결 관리 -- `/v1/embeddings`를 통한 임베딩 생성(6개 공급자, 9개 모델) -- `/v1/images/ Generations`를 통한 이미지 생성(4개 공급자, 9개 모델) -- 추론 모델을 위한 Think 태그 구문 분석(`...`) -- 엄격한 OpenAI SDK 호환성을 위한 응답 삭제 -- 제공자 간 호환성을 위한 역할 정규화(개발자→시스템, 시스템→사용자) -- 구조화된 출력 변환(json_schema → Gemini responseSchema) -- 공급자, 키, 별칭, 콤보, 설정, 가격에 대한 로컬 지속성 +## Executive Summary + +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. + +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing - Usage/cost tracking and request logging -- 다중 장치/상태 동기화를 위한 선택적 클라우드 동기화 -- API 접근 제어를 위한 IP 허용 목록/차단 목록 -- 생각하는 예산 관리(패스스루/자동/커스텀/적응형) -- 글로벌 시스템 신속한 주입 -- 세션 추적 및 지문 채취 -- 제공자별 프로필을 통해 계정당 강화된 속도 제한 -- 공급자 탄력성을 위한 회로 차단기 패턴 -- 뮤텍스 잠금을 통한 천둥 방지 무리 보호 -- 서명 기반 요청 중복 제거 캐시 -- 도메인 레이어: 모델 가용성, 비용 규칙, 대체 정책, 잠금 정책 -- 도메인 상태 지속성(폴백, 예산, 잠금, 회로 차단기를 위한 SQLite 연속 쓰기 캐시) -- 중앙화된 요청 평가를 위한 정책 엔진(잠금 → 예산 → 대체) -- p50/p95/p99 대기 시간 집계를 통한 원격 측정 요청 -- 종단 간 추적을 위한 상관 ID(X-Request-Id) -- API 키별로 옵트아웃이 가능한 규정 준수 감사 로깅 -- LLM 품질 보증을 위한 평가 프레임워크 -- 실시간 회로 차단기 상태가 포함된 탄력성 UI 대시보드 -- 모듈형 OAuth 제공자(`src/lib/oauth/providers/` 아래의 개별 모듈 12개) +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) -기본 런타임 모델: +Primary runtime model: -- `src/app/api/*` 아래의 Next.js 앱 경로는 대시보드 API와 호환성 API를 모두 구현합니다. -- `src/sse/*` + `open-sse/*`의 공유 SSE/라우팅 코어는 공급자 실행, 변환, 스트리밍, 대체 및 사용을 처리합니다.## Scope and Boundaries +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- 로컬 게이트웨이 런타임 -- 대시보드 관리 API -- 공급자 인증 및 토큰 새로 고침 -- 번역 및 SSE 스트리밍 요청 -- 로컬 상태 + 사용 지속성 -- 선택적인 클라우드 동기화 조정### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- `NEXT_PUBLIC_CLOUD_URL` 뒤에 클라우드 서비스 구현 -- 로컬 프로세스 외부의 공급자 SLA/제어 평면 -- 외부 CLI 바이너리 자체(Claude CLI, Codex CLI 등)## Dashboard Surface (Current) +### Out of Scope -`src/app/(dashboard)/dashboard/` 아래의 기본 페이지: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — 빠른 시작 + 공급자 개요 -- `/dashboard/endpoint` — 엔드포인트 프록시 + MCP + A2A + API 엔드포인트 탭 -- `/dashboard/providers` — 공급자 연결 및 자격 증명 -- `/dashboard/combos` — 콤보 전략, 템플릿, 모델 라우팅 규칙 -- `/dashboard/costs` — 비용 집계 및 가격 가시성 -- `/dashboard/analytics` — 사용 분석 및 평가 -- `/dashboard/limits` — 할당량/비율 제어 -- `/dashboard/cli-tools` — CLI 온보딩, 런타임 감지, 구성 생성 -- `/dashboard/agents` — 감지된 ACP 에이전트 + 사용자 지정 에이전트 등록 -- `/dashboard/media` — 이미지/비디오/음악 놀이터 -- `/dashboard/search-tools` — 검색 공급자 테스트 및 기록 -- `/dashboard/health` — 가동 시간, 회로 차단기, 속도 제한 -- `/dashboard/logs` — 요청/프록시/감사/콘솔 로그 -- `/dashboard/settings` — 시스템 설정 탭(일반, 라우팅, 콤보 기본값 등) -- `/dashboard/api-manager` — API 키 수명 주기 및 모델 권한## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,135 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -주요 디렉토리: +Main directories: -- 호환성 API를 위한 `src/app/api/v1/*` 및 `src/app/api/v1beta/*` -- 관리/구성 API용 `src/app/api/*` -- 다음은 `next.config.mjs` 맵 `/v1/*`을 `/api/v1/*`로 다시 작성합니다. +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -중요한 호환성 경로: +Important compatibility routes: -- `src/app/api/v1/chat/completions/route.ts` -`src/app/api/v1/messages/route.ts` -`src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` — `custom: true`를 사용하는 사용자 정의 모델을 포함합니다. -- `src/app/api/v1/embeddings/route.ts` — 임베딩 생성(6개 공급자) -- `src/app/api/v1/images/세대/route.ts` — 이미지 생성(Antigravity/Nebius를 포함한 4개 이상의 공급자) -`src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — 제공자별 전용 채팅 -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — 제공자별 전용 임베딩 -- `src/app/api/v1/providers/[provider]/images/ Generations/route.ts` — 제공자별 전용 이미지 -`src/app/api/v1beta/models/route.ts` -- `src/app/api/v1beta/models/[...경로]/route.ts` +- `src/app/api/v1/chat/completions/route.ts` +- `src/app/api/v1/messages/route.ts` +- `src/app/api/v1/responses/route.ts` +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) +- `src/app/api/v1/messages/count_tokens/route.ts` +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images +- `src/app/api/v1beta/models/route.ts` +- `src/app/api/v1beta/models/[...path]/route.ts` -관리 도메인: +Management domains: -- 인증/설정: `src/app/api/auth/*`, `src/app/api/settings/*` -- 공급자/연결: `src/app/api/providers*` -- 공급자 노드: `src/app/api/provider-nodes*` -- 사용자 정의 모델: `src/app/api/provider-models`(GET/POST/DELETE) -- 모델 카탈로그: `src/app/api/models/route.ts`(GET) -- 프록시 구성: `src/app/api/settings/proxy`(GET/PUT/DELETE) + `src/app/api/settings/proxy/test`(POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- 키/별칭/콤보/가격: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- 사용법: `src/app/api/usage/*` -- 동기화/클라우드: `src/app/api/sync/*`, `src/app/api/cloud/*` -- CLI 도구 도우미: `src/app/api/cli-tools/*` -- IP 필터: `src/app/api/settings/ip-filter`(GET/PUT) -- 생각하는 예산: `src/app/api/settings/thinking-budget` (GET/PUT) -- 시스템 프롬프트: `src/app/api/settings/system-prompt`(GET/PUT) -- 세션: `src/app/api/sessions`(GET) -- 속도 제한: `src/app/api/rate-limits`(GET) -- 복원력: `src/app/api/resilience`(GET/PATCH) — 공급자 프로필, 회로 차단기, 속도 제한 상태 -- 복원력 재설정: `src/app/api/resilience/reset`(POST) — 재설정 차단기 + 재사용 대기시간 -- 캐시 통계: `src/app/api/cache/stats`(GET/DELETE) -- 모델 가용성: `src/app/api/models/availability`(GET/POST) -- 원격 측정: `src/app/api/telemetry/summary`(GET) -- 예산: `src/app/api/usage/budget`(GET/POST) -- 대체 체인: `src/app/api/fallback/chains`(GET/POST/DELETE) -- 규정 준수 감사: `src/app/api/compliance/audit-log`(GET) -- 평가: `src/app/api/evals`(GET/POST), `src/app/api/evals/[suiteId]`(GET) -- 정책: `src/app/api/policies`(GET/POST)## 2) SSE + Translation Core +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -주요 흐름 모듈: +## 2) SSE + Translation Core -- 항목: `src/sse/handlers/chat.ts` -- 핵심 오케스트레이션: `open-sse/handlers/chatCore.ts` -- 공급자 실행 어댑터: `open-sse/executors/*` -- 형식 감지/공급자 구성: `open-sse/services/provider.ts` -- 모델 구문 분석/해결: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- 계정 대체 논리: `open-sse/services/accountFallback.ts` -- 번역 레지스트리: `open-sse/translator/index.ts` -- 스트림 변환: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- 사용량 추출/정규화: `open-sse/utils/usageTracking.ts` -- Think 태그 파서: `open-sse/utils/thinkTagParser.ts` -- 임베딩 핸들러: `open-sse/handlers/embeddings.ts` -- 포함 공급자 레지스트리: `open-sse/config/embeddingRegistry.ts` -- 이미지 생성 핸들러: `open-sse/handlers/imageGeneration.ts` -- 이미지 제공자 레지스트리: `open-sse/config/imageRegistry.ts` -- 응답 삭제: `open-sse/handlers/responseSanitizer.ts` -- 역할 정규화: `open-sse/services/roleNormalizer.ts` +Main flow modules: -서비스(비즈니스 로직): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- 계정 선택/점수: `open-sse/services/accountSelector.ts` -- 컨텍스트 수명주기 관리: `open-sse/services/contextManager.ts` -- IP 필터 적용: `open-sse/services/ipFilter.ts` -- 세션 추적: `open-sse/services/sessionManager.ts` -- 중복 제거 요청: `open-sse/services/signatureCache.ts` -- 시스템 프롬프트 주입: `open-sse/services/systemPrompt.ts` -- 생각하는 예산 관리: `open-sse/services/thinkingBudget.ts` -- 와일드카드 모델 라우팅: `open-sse/services/wildcardRouter.ts` -- 비율 제한 관리: `open-sse/services/rateLimitManager.ts` -- 회로 차단기: `open-sse/services/circuitBreaker.ts` +Services (business logic): -도메인 레이어 모듈: +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- 모델 가용성: `src/lib/domain/modelAvailability.ts` -- 비용 규칙/예산: `src/lib/domain/costRules.ts` -- 대체 정책: `src/lib/domain/fallbackPolicy.ts` -- 콤보 리졸버: `src/lib/domain/comboResolver.ts` -- 잠금 정책: `src/lib/domain/lockoutPolicy.ts` -- 정책 엔진: `src/domain/policyEngine.ts` — 중앙 집중식 잠금 → 예산 → 대체 평가 -- 오류 코드 카탈로그: `src/lib/domain/errorCodes.ts` -- 요청 ID: `src/lib/domain/requestId.ts` -- 가져오기 시간 초과: `src/lib/domain/fetchTimeout.ts` -- 원격 측정 요청: `src/lib/domain/requestTelemetry.ts` -- 규정 준수/감사: `src/lib/domain/compliance/index.ts` -- 평가 실행기: `src/lib/domain/evalRunner.ts` -- 도메인 상태 지속성: `src/lib/db/domainState.ts` — 폴백 체인, 예산, 비용 기록, 잠금 상태, 회로 차단기를 위한 SQLite CRUD +Domain layer modules: -OAuth 제공자 모듈(`src/lib/oauth/providers/` 아래의 개별 파일 12개): +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -- 레지스트리 인덱스: `src/lib/oauth/providers/index.ts` -- 개별 제공자: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` -- 씬 래퍼: `src/lib/oauth/providers.ts` — 개별 모듈에서 다시 내보내기## 3) Persistence Layer +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -기본 상태 DB(SQLite): +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -- 핵심 인프라: `src/lib/db/core.ts`(better-sqlite3, 마이그레이션, WAL) -- 파사드 다시 내보내기: `src/lib/localDb.ts`(호출자를 위한 얇은 호환성 레이어) -- 파일: `${DATA_DIR}/storage.sqlite`(또는 설정된 경우 `$XDG_CONFIG_HOME/omniroute/storage.sqlite`, 그렇지 않으면 `~/.omniroute/storage.sqlite`) -- 엔터티(테이블 + KV 네임스페이스): 공급자 연결, 공급자 노드, 모델 별칭, 콤보, apiKeys, 설정, 가격 책정,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +## 3) Persistence Layer -사용 지속성: +Primary state DB (SQLite): -- 외관: `src/lib/usageDb.ts`(`src/lib/usage/*`에서 분해된 모듈) -- `storage.sqlite`의 SQLite 테이블: `usage_history`, `call_logs`, `proxy_logs` -- 호환성/디버그를 위해 선택적 파일 아티팩트가 남아 있습니다(`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- 레거시 JSON 파일이 있는 경우 시작 마이그레이션을 통해 SQLite로 마이그레이션됩니다. +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -도메인 상태 DB(SQLite): +Usage persistence: -- `src/lib/db/domainState.ts` — 도메인 상태에 대한 CRUD 작업 -- 테이블(`src/lib/db/core.ts`에서 생성됨): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- 연속 쓰기 캐시 패턴: 메모리 내 맵은 런타임 시 권한을 갖습니다. 변이는 SQLite에 동기적으로 기록됩니다. 콜드 스타트 ​​시 DB에서 상태가 복원됩니다.## 4) Auth + Security Surfaces +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- 대시보드 쿠키 인증: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- API 키 생성/검증: `src/shared/utils/apiKey.ts` -- 'providerConnections' 항목에 유지되는 공급자 비밀 -- `open-sse/utils/proxyFetch.ts`(env vars) 및 `open-sse/utils/networkProxy.ts`(공급자별로 구성 가능 또는 전역)를 통한 아웃바운드 프록시 지원## 5) Cloud Sync +Domain State DB (SQLite): -- 스케줄러 초기화: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- 정기 작업: `src/shared/services/cloudSyncScheduler.ts` -- 주기적 작업: `src/shared/services/modelSyncScheduler.ts` -- 제어 경로: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -334,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -폴백 결정은 상태 코드와 오류 메시지 경험적 방법을 사용하는 'open-sse/services/accountFallback.ts'에 의해 이루어집니다. 콤보 라우팅은 하나의 추가 보호 기능을 추가합니다. 업스트림 콘텐츠 블록 및 역할 검증 실패와 같은 공급자 범위 400은 모델 로컬 실패로 처리되므로 이후 콤보 대상이 계속 실행될 수 있습니다.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -364,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -실시간 트래픽 중 새로 고침은 실행기 `refreshCredentials()`를 통해 `open-sse/handlers/chatCore.ts` 내에서 실행됩니다.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -396,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -클라우드가 활성화되면 `CloudSyncScheduler`에 의해 주기적 동기화가 트리거됩니다.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -497,12 +532,14 @@ erDiagram } ``` -물리적 저장 파일: +Physical storage files: -- 기본 런타임 DB: `${DATA_DIR}/storage.sqlite` -- 요청 로그 줄: `${DATA_DIR}/log.txt`(compat/debug 아티팩트) -- 구조화된 호출 페이로드 아카이브: `${DATA_DIR}/call_logs/` -- 선택적 변환기/요청 디버그 세션: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -537,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: 호환성 API -- `src/app/api/v1/providers/[provider]/*`: 제공자별 전용 경로(채팅, 임베딩, 이미지) -- `src/app/api/providers*`: 공급자 CRUD, 유효성 검사, 테스트 -- `src/app/api/provider-nodes*`: 맞춤형 호환 노드 관리 -- `src/app/api/provider-models`: 사용자 정의 모델 관리(CRUD) -- `src/app/api/models/route.ts`: 모델 카탈로그 API(별칭 + 사용자 정의 모델) -- `src/app/api/oauth/*`: OAuth/장치 코드 흐름 -- `src/app/api/keys*`: 로컬 API 키 수명 주기 -- `src/app/api/models/alias`: 별칭 관리 -- `src/app/api/combos*`: 대체 콤보 관리 -- `src/app/api/pricing`: 비용 계산을 위한 가격 재정의 -- `src/app/api/settings/proxy`: 프록시 구성(GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: 아웃바운드 프록시 연결 테스트(POST) -- `src/app/api/usage/*`: 사용 및 로그 API -- `src/app/api/sync/*` + `src/app/api/cloud/*`: 클라우드 동기화 및 클라우드 연결 도우미 -- `src/app/api/cli-tools/*`: 로컬 CLI 구성 작성자/검사기 -- `src/app/api/settings/ip-filter`: IP 허용 목록/차단 목록(GET/PUT) -- `src/app/api/settings/thinking-budget`: Thinking Token 예산 구성(GET/PUT) -- `src/app/api/settings/system-prompt`: 전역 시스템 프롬프트(GET/PUT) -- `src/app/api/sessions`: 활성 세션 목록(GET) -- `src/app/api/rate-limits`: 계정별 비율 제한 상태(GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: 요청 구문 분석, 콤보 처리, 계정 선택 루프 -- `open-sse/handlers/chatCore.ts`: 변환, 실행기 디스패치, 재시도/새로 고침 처리, 스트림 설정 -- `open-sse/executors/*`: 공급자별 네트워크 및 형식 동작### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: 번역기 레지스트리 및 오케스트레이션 -- 번역자 요청: `open-sse/translator/request/*` -- 응답 번역기: `open-sse/translator/response/*` -- 형식 상수: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: SQLite의 지속적인 구성/상태 및 도메인 지속성 -- `src/lib/localDb.ts`: DB 모듈에 대한 호환성 다시 내보내기 -- `src/lib/usageDb.ts`: SQLite 테이블 위에 있는 사용 내역/호출 로그 파사드## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -각 공급자에는 URL 구축, 헤더 구성, 지수 백오프를 사용한 재시도, 자격 증명 새로 고침 후크 및 `execute()` 조정 방법을 제공하는 `BaseExecutor`(`open-sse/executors/base.ts`에 있음)를 확장하는 특수 실행기가 있습니다. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| 집행자 | 공급자 | 특수취급 | -| ------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------- | -| `기본 실행자` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | 공급자별 동적 URL/헤더 구성 | -| '반중력실행자' | 구글 반중력 | 사용자 정의 프로젝트/세션 ID, 구문 분석 후 재시도 | -| `CodexExecutor` | OpenAI 코덱스 | 시스템 지침을 주입하고 추론 노력을 강요 | -| `CursorExecutor` | 커서 IDE | ConnectRPC 프로토콜, Protobuf 인코딩, 체크섬을 통한 서명 요청 | -| `GithubExecutor` | GitHub 부조종사 | Copilot 토큰 새로 고침, VSCode 모방 헤더 | -| '키로집행자' | AWS 코드위스퍼러/키로 | AWS EventStream 바이너리 형식 → SSE 변환 | -| `GeminiCLIExecutor` | 제미니 CLI | Google OAuth 토큰 새로고침 주기 | +### Persistence -다른 모든 공급자(사용자 정의 호환 노드 포함)는 `DefaultExecutor`를 사용합니다.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| 공급자 | 형식 | 인증 | 스트림 | 비스트림 | 토큰 새로고침 | 사용 API | -| ----------------- | -------------- | -------------------- | ----------------- | -------- | ------------- | ------------------ | ------------------------------ | -| 클로드 | 클로드 | API 키/OAuth | ✅ | ✅ | ✅ | ⚠️ 관리자 전용 | -| 쌍둥이자리 | 쌍둥이자리 | API 키/OAuth | ✅ | ✅ | ✅ | ⚠️ 클라우드 콘솔 | -| 제미니 CLI | 쌍둥이자리 CLI | OAuth | ✅ | ✅ | ✅ | ⚠️ 클라우드 콘솔 | -| 반중력 | 반중력 | OAuth | ✅ | ✅ | ✅ | ✅ 전체 할당량 API | -| 오픈AI | 공개 | API 키 | ✅ | ✅ | ❌ | ❌ | -| 코덱스 | openai-응답 | OAuth | ✅ 강제 | ❌ | ✅ | ✅ 비율 제한 | -| GitHub 부조종사 | 공개 | OAuth + Copilot 토큰 | ✅ | ✅ | ✅ | ✅ 할당량 스냅샷 | -| 커서 | 커서 | 사용자 정의 체크섬 | ✅ | ✅ | ❌ | ❌ | -| 키로 | 키로 | AWS SSO OIDC | ✅ (이벤트스트림) | ❌ | ✅ | ✅ 사용 제한 | -| 퀀 | 공개 | OAuth | ✅ | ✅ | ✅ | ⚠️ 요청에 따라 | -| Qoder | 공개 | OAuth(기본) | ✅ | ✅ | ✅ | ⚠️ 요청에 따라 | -| 오픈라우터 | 공개 | API 키 | ✅ | ✅ | ❌ | ❌ | -| GLM/키미/미니맥스 | 클로드 | API 키 | ✅ | ✅ | ❌ | ❌ | -| 딥시크 | 공개 | API 키 | ✅ | ✅ | ❌ | ❌ | -| 그로크 | 공개 | API 키 | ✅ | ✅ | ❌ | ❌ | -| xAI(그록) | 공개 | API 키 | ✅ | ✅ | ❌ | ❌ | -| 미스트랄 | 공개 | API 키 | ✅ | ✅ | ❌ | ❌ | -| 당혹감 | 공개 | API 키 | ✅ | ✅ | ❌ | ❌ | -| 함께하는 AI | 공개 | API 키 | ✅ | ✅ | ❌ | ❌ | -| 불꽃놀이 AI | 공개 | API 키 | ✅ | ✅ | ❌ | ❌ | -| 대뇌 | 공개 | API 키 | ✅ | ✅ | ❌ | ❌ | -| 코히어 | 공개 | API 키 | ✅ | ✅ | ❌ | ❌ | -| 엔비디아 NIM | 공개 | API 키 | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -감지된 소스 형식은 다음과 같습니다. +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. -- 'openai' -`openai-응답` -- '클로드' -- `쌍둥이자리` +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | -대상 형식은 다음과 같습니다. +All other providers (including custom compatible nodes) use the `DefaultExecutor`. -- OpenAI 채팅/응답 -- 클로드 -- Gemini/Gemini-CLI/반중력 봉투 -- 키로 -- 커서 +## Provider Compatibility Matrix -번역에서는**OpenAI를 허브 형식**으로 사용합니다. 모든 변환은 중간 형식으로 OpenAI를 거칩니다.``` +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: + +- `openai` +- `openai-responses` +- `claude` +- `gemini` + +Target formats include: + +- OpenAI chat/Responses +- Claude +- Gemini/Gemini-CLI/Antigravity envelope +- Kiro +- Cursor + +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -번역은 소스 페이로드 형태와 공급자 대상 형식에 따라 동적으로 선택됩니다. +Additional processing layers in the translation pipeline: -번역 파이프라인의 추가 처리 계층: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**응답 삭제**— OpenAI 형식 응답(스트리밍 및 비스트리밍 모두)에서 비표준 필드를 제거하여 엄격한 SDK 규정 준수를 보장합니다. --**역할 정규화**— OpenAI가 아닌 대상에 대해 `개발자` → `시스템`을 변환합니다. 시스템 역할을 거부하는 모델의 경우 `system` → `user`를 병합합니다(GLM, ERNIE) --**Think 태그 추출**— 콘텐츠의 `...` 블록을 `reasoning_content` 필드로 구문 분석합니다. --**구조화된 출력**— OpenAI `response_format.json_schema`를 Gemini의 `responseMimeType` + `responseSchema`로 변환합니다.## Supported API Endpoints +## Supported API Endpoints -| 엔드포인트 | 형식 | 핸들러 | -| ------------------------------------- | ------------------ | ------------------------------------------------------ | -| `POST /v1/chat/completions` | OpenAI 채팅 | `src/sse/handlers/chat.ts` | -| `POST /v1/messages` | 클로드 메시지 | 동일한 핸들러(자동 감지) | -| `POST /v1/응답` | OpenAI 응답 | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/임베딩` | OpenAI 임베딩 | `open-sse/handlers/embeddings.ts` | -| `GET /v1/embeddings` | 모델 목록 | API 경로 | -| `POST /v1/이미지/세대` | OpenAI 이미지 | `open-sse/handlers/imageGeneration.ts` | -| `GET /v1/이미지/세대` | 모델 목록 | API 경로 | -| `POST /v1/providers/{provider}/chat/completions` | OpenAI 채팅 | 모델 검증을 통한 제공자별 전용 | -| `POST /v1/providers/{provider}/embeddings` | OpenAI 임베딩 | 모델 검증을 통한 제공자별 전용 | -| `POST /v1/providers/{provider}/이미지/세대` | OpenAI 이미지 | 모델 검증을 통한 제공자별 전용 | -| `POST /v1/messages/count_tokens` | 클로드 토큰 개수 | API 경로 | -| `GET /v1/models` | OpenAI 모델 목록 | API 경로(채팅 + 임베딩 + 이미지 + 사용자 정의 모델) | -| `GET /api/models/catalog` | 카탈로그 | 공급자 + 유형별로 그룹화된 모든 모델 | -| `POST /v1beta/models/*:streamGenerateContent` | 쌍둥이 자리 원주민 | API 경로 | -| `GET/PUT/DELETE /api/settings/proxy` | 프록시 구성 | 네트워크 프록시 구성 | -| `POST /api/settings/proxy/test` | 프록시 연결 | 프록시 상태/연결 테스트 엔드포인트 | -| `GET/POST/DELETE /api/provider-models` | 공급자 모델 | 사용자 정의 및 관리형 사용 가능한 모델을 지원하는 공급자 모델 메타데이터 |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -우회 핸들러(`open-sse/utils/bypassHandler.ts`)는 Claude CLI의 알려진 "일시적" 요청(예열 핑, 타이틀 추출 및 토큰 카운트)을 가로채고 업스트림 공급자 토큰을 사용하지 않고**가짜 응답**을 반환합니다. 이는 `User-Agent`에 `claude-cli`가 포함된 경우에만 트리거됩니다.## Request Logger Pipeline +## Bypass Handler -요청 로거(`open-sse/utils/requestLogger.ts`)는 기본적으로 비활성화되고 `ENABLE_REQUEST_LOGS=true`를 통해 활성화되는 7단계 디버그 로깅 파이프라인을 제공합니다.``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -각 요청 세션마다 `/logs//`에 파일이 기록됩니다.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- 일시적/속도/인증 오류에 대한 공급자 계정 쿨다운 -- 요청 실패 전 계정 대체 -- 현재 모델/공급자 경로가 소진되면 콤보 모델 대체## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- 새로 고칠 수 있는 공급자에 대한 사전 확인 및 재시도를 통한 새로 고침 -- 코어 경로에서 새로 고침 시도 후 401/403 재시도## 3) Stream Safety +## 2) Token Expiry -- 연결 해제 인식 스트림 컨트롤러 -- 스트림 끝 플러시 및 `[DONE]` 처리가 포함된 번역 스트림 -- 공급자 사용량 메타데이터가 누락된 경우 사용량 추정 대체## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- 동기화 오류가 표시되지만 로컬 런타임은 계속됩니다. -- 스케줄러에는 재시도 가능 논리가 있지만 주기적인 실행은 현재 기본적으로 단일 시도 동기화를 호출합니다.## 5) Data Integrity +## 3) Stream Safety -- 시작 시 SQLite 스키마 마이그레이션 및 자동 업그레이드 후크 -- 레거시 JSON → SQLite 마이그레이션 호환성 경로## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -런타임 가시성 소스: +## 4) Cloud Sync Degradation --`src/sse/utils/logger.ts`의 콘솔 로그 +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -- SQLite의 요청별 사용량 집계(`usage_history`, `call_logs`, `proxy_logs`) -- `settings.detailed_logs_enabled=true`인 경우 SQLite(`request_detail_logs`)에서 4단계 상세 페이로드 캡처 -- `log.txt`의 텍스트 요청 상태 로그(선택 사항/호환) -- `ENABLE_REQUEST_LOGS=true`인 경우 `logs/` 아래의 선택적 심층 요청/변환 로그 -- UI 소비를 위한 대시보드 사용 엔드포인트(`/api/usage/*`) +## 5) Data Integrity -세부 요청 페이로드 캡처는 라우팅된 호출당 최대 4개의 JSON 페이로드 단계를 저장합니다. +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- 클라이언트로부터 받은 원시 요청 -- 번역된 요청이 실제로 업스트림으로 전송됨 -- 공급자 응답이 JSON으로 재구성되었습니다. 스트리밍된 응답은 최종 요약과 스트림 메타데이터로 압축됩니다. -- OmniRoute에서 반환된 최종 클라이언트 응답 스트리밍된 응답은 동일한 압축 요약 형식으로 저장됩니다.## Security-Sensitive Boundaries +## Observability and Operational Signals -- JWT 비밀(`JWT_SECRET`)은 대시보드 세션 쿠키 확인/서명을 보호합니다. -- 최초 실행 프로비저닝을 위해 초기 비밀번호 부트스트랩(`INITIAL_PASSWORD`)을 명시적으로 구성해야 합니다. -- API 키 HMAC 비밀(`API_KEY_SECRET`)은 생성된 로컬 API 키 형식을 보호합니다. -- 공급자 비밀(API 키/토큰)은 로컬 DB에 유지되며 파일 시스템 수준에서 보호되어야 합니다. -- 클라우드 동기화 엔드포인트는 API 키 인증 + 머신 ID 의미 체계를 사용합니다.## Environment and Runtime Matrix +Runtime visibility sources: -코드에서 적극적으로 사용되는 환경 변수: +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -- 앱/인증: `JWT_SECRET`, `INITIAL_PASSWORD` -- 저장공간: `DATA_DIR` -- 호환 노드 동작: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- 선택적 저장소 기반 재정의(`DATA_DIR`이 설정되지 않은 경우 Linux/macOS): `XDG_CONFIG_HOME` -- 보안 해싱: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- 로깅: `ENABLE_REQUEST_LOGS` -- 동기화/클라우드 URL 지정: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- 아웃바운드 프록시: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` 및 소문자 변형 -- SOCKS5 기능 플래그: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- 플랫폼/런타임 도우미(앱별 구성 아님): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +Detailed request payload capture stores up to four JSON payload stages per routed call: -1. `usageDb` 및 `localDb`는 레거시 파일 마이그레이션과 동일한 기본 디렉터리 정책(`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`)을 공유합니다. -2. 의미적 드리프트를 피하기 위해 `/api/v1/route.ts`는 `/api/v1/models`(`src/app/api/v1/models/catalog.ts`)에서 사용되는 동일한 통합 카탈로그 빌더에 위임합니다. -3. 요청 로거가 활성화되면 전체 헤더/본문을 씁니다. 로그 디렉토리를 중요하게 취급하십시오. -4. 클라우드 동작은 올바른 'NEXT_PUBLIC_BASE_URL' 및 클라우드 엔드포인트 연결 가능성에 따라 달라집니다. -5. `open-sse/` 디렉터리는 `@omniroute/open-sse`**npm 작업 공간 패키지**로 게시됩니다. 소스 코드는 `@omniroute/open-sse/...`(Next.js `transpilePackages`로 해결됨)를 통해 이를 가져옵니다. 이 문서의 파일 경로는 일관성을 위해 여전히 `open-sse/` 디렉토리 이름을 사용합니다. -6. 대시보드의 차트는 액세스 가능한 대화형 분석 시각화(모델 사용량 막대 차트, 성공률이 포함된 공급자 분석 테이블)를 위해**Recharts**(SVG 기반)를 사용합니다. -7. E2E 테스트는**Playwright**(`tests/e2e/`)를 사용하고 `npm run test:e2e`를 통해 실행됩니다. 단위 테스트는**Node.js 테스트 실행기**(`tests/unit/`)를 사용하고 `npm run test:unit`을 통해 실행됩니다. `src/` 아래의 소스 코드는**TypeScript**(`.ts`/`.tsx`)입니다. `open-sse/` 작업 공간은 JavaScript(`.js`)로 유지됩니다. -8. 설정 페이지는 보안, 라우팅(6개의 전역 전략: 채우기 우선, 라운드 로빈, p2c, 무작위, 최소 사용, 비용 최적화), 탄력성(편집 가능한 속도 제한, 회로 차단기, 정책), AI(생각 예산, 시스템 프롬프트, 프롬프트 캐시), 고급(프록시)의 5개 탭으로 구성됩니다.## Operational Verification Checklist +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form -- 소스에서 빌드: `npm run build` -- Docker 이미지 빌드: `docker build -t omniroute .` -- 서비스 시작 및 확인: +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: - `GET /api/settings` - `GET /api/v1/models` -- `PORT=20128`인 경우 CLI 대상 기본 URL은 `http://:20128/v1`이어야 합니다. +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/ko/docs/FEATURES.md b/docs/i18n/ko/docs/FEATURES.md index 3f6d5fc577..5a47cc09ed 100644 --- a/docs/i18n/ko/docs/FEATURES.md +++ b/docs/i18n/ko/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -OmniRoute 대시보드의 모든 섹션에 대한 시각적 가이드입니다.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -AI 공급자 연결 관리: OAuth 공급자(Claude Code, Codex, Gemini CLI), API 키 공급자(Groq, DeepSeek, OpenRouter) 및 무료 공급자(Qoder, Qwen, Kiro). Kiro 계정에는 크레딧 잔액 추적 기능이 포함되어 있습니다. 남은 크레딧, 총 허용량, 갱신 날짜는 대시보드 → 사용량에서 확인할 수 있습니다.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -우선 순위, 가중치 적용, 라운드 로빈, 무작위, 최소 사용, 비용 최적화 등 6가지 전략을 사용하여 모델 라우팅 콤보를 만듭니다. 각 콤보는 자동 폴백을 통해 여러 모델을 연결하며 빠른 템플릿과 준비 상태 확인을 포함합니다.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -토큰 소비, 비용 추정, 활동 히트맵, 주간 분포 차트 및 공급자별 분석을 포함한 포괄적인 사용량 분석입니다.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -실시간 모니터링: 가동 시간, 메모리, 버전, 대기 시간 백분위수(p50/p95/p99), 캐시 통계 및 공급자 회로 차단기 상태.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -API 번역 디버깅을 위한 4가지 모드:**플레이그라운드**(형식 변환기),**채팅 테스터**(실시간 요청),**테스트 벤치**(일괄 테스트),**라이브 모니터**(실시간 스트림).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -대시보드에서 직접 모델을 테스트해 보세요. 공급자, 모델 및 엔드포인트를 선택하고, Monaco Editor로 프롬프트를 작성하고, 실시간으로 응답을 스트리밍하고, 중간 스트림을 중단하고, 타이밍 측정항목을 확인하세요.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -전체 대시보드에 대한 사용자 정의 가능한 색상 테마입니다. 7가지 사전 설정된 색상(산호색, 파란색, 빨간색, 녹색, 보라색, 주황색, 청록색) 중에서 선택하거나 16진수 색상을 선택하여 사용자 정의 테마를 만드세요. 밝음, 어두움 및 시스템 모드를 지원합니다.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -탭이 포함된 종합 설정 패널: +Comprehensive settings panel with tabs: --**일반**— 시스템 스토리지, 백업 관리(데이터베이스 내보내기/가져오기) -**모양**— 테마 선택기(어두움/밝음/시스템), 색상 테마 사전 설정 및 사용자 정의 색상, 상태 로그 표시, 사이드바 항목 표시 제어 -**보안**— API 엔드포인트 보호, 맞춤형 공급자 차단, IP 필터링, 세션 정보 -**라우팅**— 모델 별칭, 백그라운드 작업 성능 저하 -**복원력**— 속도 제한 지속성, 회로 차단기 조정, 금지된 계정 자동 비활성화, 공급자 만료 모니터링 -**고급**— 구성 재정의, 구성 감사 추적, 대체 성능 저하 모드![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -AI 코딩 도구에 대한 원클릭 구성: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor 및 Factory Droid. 자동화된 구성 적용/재설정, 연결 프로필 및 모델 매핑 기능이 있습니다.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -CLI 에이전트를 검색하고 관리하기 위한 대시보드입니다. 다음을 포함하는 14개의 내장 에이전트(Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp)의 그리드를 표시합니다. +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**설치 상태**— 버전 감지를 통해 설치됨/찾을 수 없음 -**프로토콜 배지**— stdio, HTTP 등 -**사용자 지정 에이전트**— 양식(이름, 바이너리, 버전 명령, 생성 인수)을 통해 모든 CLI 도구 등록 -**CLI 지문 일치**— 기본 CLI 요청 서명과 일치하도록 공급자별 토글을 통해 프록시 IP를 유지하면서 금지 위험을 줄입니다.--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -대시보드에서 이미지, 비디오, 음악을 생성하세요. OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open 및 MusicGen을 지원합니다.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -공급자, 모델, 계정 및 API 키별로 필터링하여 실시간 요청 로깅. 상태 코드, 토큰 사용량, 대기 시간 및 응답 세부 정보를 표시합니다.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -기능 분석이 포함된 통합 API 엔드포인트: 채팅 완료, 응답 API, 임베딩, 이미지 생성, 순위 재지정, 오디오 전사, 텍스트 음성 변환, 조정 및 등록된 API 키. 원격 액세스를 위한 Cloudflare Quick Tunnel 통합 및 클라우드 프록시 지원.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -API 키를 생성, 범위 지정, 취소합니다. 각 키는 전체 액세스 또는 읽기 전용 권한이 있는 특정 모델/공급업체로 제한될 수 있습니다. 사용 추적을 통한 시각적 키 관리.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -작업 유형, 행위자, 대상, IP 주소 및 타임스탬프를 기준으로 필터링하여 관리 작업을 추적합니다. 전체 보안 이벤트 내역.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Windows, macOS, Linux용 기본 Electron 데스크톱 앱입니다. 시스템 트레이 통합, 오프라인 지원, 자동 업데이트 및 원클릭 설치 기능을 갖춘 독립형 애플리케이션으로 OmniRoute를 실행하세요. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -주요 기능: +Key features: -- 서버 준비 폴링(콜드 스타트 시 빈 화면 없음) -- 포트 관리 기능이 있는 시스템 트레이 -- 콘텐츠 보안 정책 -- 단일 인스턴스 잠금 -- 재시작 시 자동 업데이트 -- 플랫폼 조건부 UI(macOS 신호등, Windows/Linux 기본 제목 표시줄) -- 강화된 Electron 빌드 패키징 — 독립 실행형 번들의 심볼릭 링크된 'node_modules'가 패키징 전에 감지 및 거부되어 빌드 시스템(v2.5.5+)에 대한 런타임 종속성을 방지합니다. +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 전체 문서는 [`electron/README.md`](../electron/README.md)를 참조하세요. +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/ko/docs/TROUBLESHOOTING.md b/docs/i18n/ko/docs/TROUBLESHOOTING.md index 036c1c38c2..513a56549e 100644 --- a/docs/i18n/ko/docs/TROUBLESHOOTING.md +++ b/docs/i18n/ko/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -OmniRoute의 일반적인 문제 및 솔루션.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| 문제 | 솔루션 | -| ----------------------------------- | --------------------------------------------------------------------------- | --- | -| 첫 번째 로그인이 작동하지 않습니다 | `.env`에서 `INITIAL_PASSWORD` 설정(하드코딩된 기본값 없음) | -| 대시보드가 ​​잘못된 포트에서 열림 | `PORT=20128` 및 `NEXT_PUBLIC_BASE_URL=http://localhost:20128` 설정 | -| `logs/` 아래에 요청 로그가 없습니다 | 'ENABLE_REQUEST_LOGS=true' 설정 | -| EACCES: 권한이 거부되었습니다 | `~/.omniroute`를 재정의하려면 `DATA_DIR=/path/to/writable/dir`을 설정하세요 | -| 라우팅 전략이 저장되지 않음 | v1.4.11+로 업데이트(설정 지속성을 위한 Zod 스키마 수정) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**원인:**공급자 할당량이 소진되었습니다. +**Cause:** Provider quota exhausted. -**수정:** +**Fix:** -1. 대시보드 할당량 추적기를 확인하세요. -2. 대체 계층과 콤보 사용 -3. 더 저렴한/무료 등급으로 전환### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**원인:**구독 할당량이 소진되었습니다. +### Rate Limiting -**수정:** +**Cause:** Subscription quota exhausted. -- 폴백 추가: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Use GLM/MiniMax as cheap backup### OAuth Token Expired +**Fix:** -OmniRoute는 토큰을 자동으로 새로 고칩니다. 문제가 지속되는 경우: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. 대시보드 → 공급자 → 재접속 -2. 공급자 연결을 삭제하고 다시 추가--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. `BASE_URL`이 실행 중인 인스턴스(예: `http://localhost:20128`)를 가리키는지 확인하세요. -2. `CLOUD_URL`이 클라우드 엔드포인트(예: `https://omniroute.dev`)를 가리키는지 확인하세요. -3. 'NEXT*PUBLIC*\*' 값을 서버측 값에 맞춰 유지하세요.### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**증상:**비스트리밍 호출에 대한 클라우드 엔드포인트의 '예기치 않은 토큰 'd'...'입니다. +### Cloud `stream=false` Returns 500 -**원인:**업스트림은 클라이언트가 JSON을 기대하는 동안 SSE 페이로드를 반환합니다. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**해결 방법:**클라우드 직접 호출에는 `stream=true`를 사용하세요. 로컬 런타임에는 SSE→JSON 대체가 포함됩니다.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. 로컬 대시보드(`/api/keys`)에서 새로운 키를 생성합니다. -2. 클라우드 동기화 실행: 클라우드 활성화 → 지금 동기화 -3. 이전/동기화되지 않은 키는 여전히 클라우드에서 '401'을 반환할 수 있습니다.--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. 런타임 필드를 확인하십시오. `curl http://localhost:20128/api/cli-tools/runtime/codex | jq' -2. 휴대용 모드의 경우: 이미지 대상 `runner-cli`(번들 CLI) 사용 -3. 호스트 마운트 모드의 경우: `CLI_EXTRA_PATHS`를 설정하고 호스트 bin 디렉터리를 읽기 전용으로 마운트합니다. -4. `installed=true` 및 `runnable=false`인 경우: 바이너리가 발견되었지만 상태 확인에 실패했습니다.### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. 대시보드 → 사용량에서 사용량 현황을 확인하세요. -2. 기본 모델을 GLM/MiniMax로 전환 -3. 중요하지 않은 작업에는 무료 계층(Gemini CLI, Qoder)을 사용하세요. -4. API 키별 비용 예산 설정: 대시보드 → API 키 → 예산--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -`.env` 파일에서 `ENABLE_REQUEST_LOGS=true`를 설정하세요. 로그는 `logs/` 디렉토리에 나타납니다.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,93 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- 기본 상태: `${DATA_DIR}/storage.sqlite`(공급자, 콤보, 별칭, 키, 설정) -- 사용법: `storage.sqlite`(`usage_history`, `call_logs`, `proxy_logs`) + 옵션 `${DATA_DIR}/log.txt` 및 `${DATA_DIR}/call_logs/`의 SQLite 테이블 -- 요청 로그: `/logs/...`(`ENABLE_REQUEST_LOGS=true`인 경우)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -공급자의 회로 차단기가 OPEN되면 대기 시간이 만료될 때까지 요청이 차단됩니다. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**수정:** +**Fix:** -1.**대시보드 → 설정 → 복원력**으로 이동합니다. 2. 영향을 받는 공급자의 회로 차단기 카드를 확인하십시오. 3.**모두 재설정**을 클릭하여 모든 차단기를 삭제하거나 쿨다운이 만료될 때까지 기다립니다. 4. 재설정하기 전에 공급자가 실제로 사용 가능한지 확인하십시오.### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -공급자가 반복적으로 OPEN 상태에 들어가는 경우: +### Provider keeps tripping the circuit breaker -1.**대시보드 → 상태 → 공급자 상태**에서 실패 패턴을 확인합니다. 2.**설정 → 탄력성 → 공급자 프로필**로 이동하여 실패 임계값을 높입니다. 3. 제공업체가 API 한도를 변경했는지 또는 재인증을 요구하는지 확인하세요. 4. 대기 시간 원격 분석 검토 - 대기 시간이 길면 시간 초과 기반 오류가 발생할 수 있습니다.--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- 올바른 접두사(`deepgram/nova-3` 또는 `assembleai/best`)를 사용하고 있는지 확인하세요. -**대시보드 → 공급자**에서 공급자가 연결되어 있는지 확인합니다.### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- 지원되는 오디오 형식 확인: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- 파일 크기가 공급자 제한(일반적으로 < 25MB) 내에 있는지 확인하세요. -- 공급자 카드에서 공급자 API 키 유효성을 확인하세요.--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -**대시보드 → 번역기**를 사용하여 형식 번역 문제를 디버깅하세요. +Use **Dashboard → Translator** to debug format translation issues: -| 모드 | 사용 시기 | -| ----------------- | ---------------------------------------------------------------------------------------- | ------------------------ | -| **놀이터** | 입력/출력 형식을 나란히 비교하세요. 실패한 요청을 붙여넣어 어떻게 변환되는지 확인하세요. | -| **채팅 테스터** | 실시간 메시지 보내기 및 헤더를 포함한 전체 요청/응답 페이로드 검사 | -| **테스트 벤치** | 형식 조합 전반에 걸쳐 일괄 테스트를 실행하여 어떤 번역이 손상되었는지 확인 | -| **라이브 모니터** | 간헐적인 번역 문제를 파악하기 위해 실시간 요청 흐름을 시청하세요 | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Thinking 태그가 표시되지 않음**— 대상 공급자가 Thinking을 지원하는지 및 Thinking 예산 설정을 확인하세요. -**도구 호출 중단**— 일부 형식 번역은 지원되지 않는 필드를 제거할 수 있습니다. 플레이그라운드 모드에서 확인 -**시스템 프롬프트 누락**— Claude와 Gemini는 시스템 프롬프트를 다르게 처리합니다. 번역 출력 확인 -**SDK는 객체 대신 원시 문자열을 반환**— v1.1.0에서 수정됨: 이제 응답 새니타이저가 OpenAI SDK Pydantic 검증 실패를 유발하는 비표준 필드(`x_groq`, `usage_breakdown` 등)를 제거합니다. -**GLM/ERNIE는 '시스템' 역할을 거부합니다**— v1.1.0에서 수정됨: 역할 정규화 프로그램이 자동으로 시스템 메시지를 호환되지 않는 모델의 사용자 메시지에 병합합니다. -**`개발자` 역할이 인식되지 않음**— v1.1.0에서 수정됨: OpenAI가 아닌 제공업체의 경우 자동으로 '시스템'으로 변환됨 -**`json_schema`가 Gemini에서 작동하지 않음**— v1.1.0에서 수정됨: `response_format`이 이제 Gemini의 `responseMimeType` + `responseSchema`로 변환됩니다.--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- 자동 비율 제한은 API 키 제공자에게만 적용됩니다(OAuth/구독 제외). -**설정 → 탄력성 → 공급자 프로필**에 자동 속도 제한이 활성화되어 있는지 확인하세요. -- 공급자가 '429' 상태 코드 또는 'Retry-After' 헤더를 반환하는지 확인하세요.### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -공급자 프로필은 다음 설정을 지원합니다. +### Tuning exponential backoff --**기본 지연**— 첫 번째 실패 후 초기 대기 시간(기본값: 1초) -**최대 지연**— 최대 대기 시간 한도(기본값: 30초) -**승수**— 연속 실패당 지연을 늘리는 정도(기본값: 2x)### Anti-thundering herd +Provider profiles support these settings: -많은 동시 요청이 속도 제한 공급자에 도달하면 OmniRoute는 뮤텍스와 자동 속도 제한을 사용하여 요청을 직렬화하고 계단식 오류를 방지합니다. API 키 제공자의 경우 이는 자동입니다.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -일부 OmniRoute 사용자는 게이트웨이를 RAG 또는 에이전트 스택 앞에 배치합니다. 이러한 설정에서는 이상한 패턴을 보는 것이 일반적입니다. OmniRoute는 정상으로 보이지만(공급자 작동, 라우팅 프로필 정상, 속도 제한 경고 없음) 최종 대답은 여전히 ​​잘못되었습니다. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -실제로 이러한 사고는 일반적으로 게이트웨이 자체가 아닌 다운스트림 RAG 파이프라인에서 발생합니다. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -이러한 실패를 설명하기 위해 공유 어휘를 원하는 경우 16개의 반복되는 RAG/LLM 실패 패턴을 정의하는 외부 MIT 라이센스 텍스트 리소스인 WFGY ProblemMap을 사용할 수 있습니다. 높은 수준에서는 다음을 다룹니다. +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- 검색 표류 및 깨진 컨텍스트 경계 -- 비어 있거나 오래된 인덱스 및 벡터 저장소 -- 임베딩 대 의미론적 불일치 -- 프롬프트 어셈블리 및 컨텍스트 창 문제 -- 논리 붕괴와 과신한 답변 -- 긴 체인 및 에이전트 조정 실패 -- 다중 에이전트 메모리 및 역할 드리프트 -- 배포 및 부트스트랩 주문 문제 +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -아이디어는 간단합니다. +The idea is simple: -1. 잘못된 응답을 조사할 때 다음을 캡처합니다. - - 사용자 작업 및 요청 - - OmniRoute의 경로 또는 공급자 콤보 - - 다운스트림에서 사용되는 모든 RAG 컨텍스트(검색된 문서, 도구 호출 등) -2. 사건을 하나 또는 두 개의 WFGY ProblemMap 번호(`No.1` ~ `No.16`)에 매핑합니다. -3. OmniRoute 로그 옆에 있는 자체 대시보드, Runbook 또는 사건 추적기에 번호를 저장합니다. -4. 해당 WFGY 페이지를 사용하여 RAG 스택, 검색기 또는 라우팅 전략을 변경해야 하는지 결정합니다. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -전문과 구체적인 레시피는 여기에 있습니다(MIT 라이센스, 텍스트만): +Full text and concrete recipes live here (MIT license, text only): -[WFGY 문제 맵 README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) +[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -OmniRoute 뒤에서 RAG 또는 에이전트 파이프라인을 실행하지 않는 경우 이 섹션을 무시할 수 있습니다.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**GitHub 문제**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**아키텍처**: 내부 세부정보는 [`docs/ARCHITECTURE.md`](ARCHITECTURE.md)를 참조하세요. -**API 참조**: 모든 엔드포인트는 [`docs/API_REFERENCE.md`](API_REFERENCE.md)를 참조하세요. -**헬스 대시보드**:**대시보드 → 헬스**에서 실시간 시스템 상태 확인 -**번역기**:**대시보드 → 번역기**를 사용하여 형식 문제 디버깅 +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/ko/llm.txt b/docs/i18n/ko/llm.txt new file mode 100644 index 0000000000..52cacdb85c --- /dev/null +++ b/docs/i18n/ko/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (한국어) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## 개요 + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### 보안 +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/ms/README.md b/docs/i18n/ms/README.md index 5fc12db277..7e60a1580e 100644 --- a/docs/i18n/ms/README.md +++ b/docs/i18n/ms/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Proksi API universal anda — satu titik akhir, 60+ pembekal, masa henti sifar. Kini dengan**Pelayan MCP (25 alatan)**,**Protokol A2A**,**Sistem Memori/Kemahiran**&**Aplikasi Desktop Elektron**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Penyelesaian Sembang • Pembenaman • Penjanaan Imej • Video • Muzik • Audio • Kedudukan Semula •**Carian Web**• Pelayan MCP • Protokol A2A • 100% TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Proksi API universal anda — satu titik akhir, 60+ pembekal, masa henti sifar. [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Tapak web](https://omniroute.online) • [🚀 Mula Pantas](#-permulaan-cepat) • [💡 Ciri](#-ciri-kunci) • [📖 Dokumen](#-dokumentasi) • [💰 Harga](#-harga-sepintas lalu) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Terdapat dalam:**🇺🇸 [Bahasa Inggeris](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipina](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,28 +60,30 @@ _Proksi API universal anda — satu titik akhir, 60+ pembekal, masa henti sifar. ## 📸 Dashboard Preview - -Klik untuk melihat tangkapan skrin papan pemuka +
+Click to see dashboard screenshots -| Halaman | Tangkapan skrin | -| ------------------ | -------------------------------------------------- | ---------- | -| **Pembekal** | ![Pembekal](docs/screenshots/01-providers.png) | -| **Kombo** | ![Kombo](docs/screenshots/02-combos.png) | -| **Analisis** | ![Analytics](docs/screenshots/03-analytics.png) | -| **Kesihatan** | ![Kesihatan](docs/screenshots/04-health.png) | -| **Penterjemah** | ![Penterjemah](docs/screenshots/05-translator.png) | -| **Tetapan** | ![Tetapan](docs/screenshots/06-settings.png) | -| **Alat CLI** | ![Alat CLI](docs/screenshots/07-cli-tools.png) | -| **Log Penggunaan** | ![Penggunaan](docs/screenshots/08-usage.png) | -| **Titik tamat** | ![Endpoints](docs/screenshots/09-endpoint.png) |
| +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | + + --- ### 🤖 Free AI Provider for your favorite coding agents -_Sambungkan mana-mana alat IDE atau CLI berkuasa AI melalui OmniRoute — get laluan API percuma untuk pengekodan tanpa had._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ - + @@ -129,8 +138,8 @@ _Sambungkan mana-mana alat IDE atau CLI berkuasa AI melalui OmniRoute — get la @@ -143,463 +152,537 @@ _Sambungkan mana-mana alat IDE atau CLI berkuasa AI melalui OmniRoute — get la - +
@@ -116,7 +125,7 @@ _Sambungkan mana-mana alat IDE atau CLI berkuasa AI melalui OmniRoute — get la OpenCode
- Kod Terbuka + OpenCode

⭐ 106K
- Kod Claude
- Kod Claude + Claude Code
+ Claude Code

⭐ 67.3K
- Kod Kilo
- Kod Kilo + Kilo Code
+ Kilo Code

⭐ 15.5K
-📡 Semua ejen menyambung melalui http://localhost:20128/v1 atau http://cloud.omniroute.online/v1 — satu konfigurasi, model tanpa had dan kuota--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Berhenti membazir wang dan mencapai had:** +**Stop wasting money and hitting limits:** -- Kuota langganan tamat tempoh tidak digunakan setiap bulan -- Had kadar menghalang anda pertengahan pengekodan -- API Mahal ($20-50/bulan setiap pembekal) -- Penukaran manual antara pembekal +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute menyelesaikan ini:** +**OmniRoute solves this:** -- ✅**Maksimumkan langganan**- Jejaki kuota, gunakan setiap bit sebelum ditetapkan semula -- ✅**Auto sandaran**- Langganan → Kunci API → Murah → Percuma, masa henti sifar -- ✅**Berbilang akaun**- Round-robin antara akaun bagi setiap pembekal -- ✅**Universal**- Berfungsi dengan Kod Claude, Codex, Gemini CLI, Kursor, Cline, OpenClaw, sebarang alat CLI--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Sertai komuniti kami!**[Kumpulan WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Dapatkan bantuan, kongsi petua dan kekal kemas kini. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Tapak web**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Isu**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Kumpulan Komuniti](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Menyumbang**: Lihat [CONTRIBUTING.md](CONTRIBUTING.md), buka PR atau pilih `isu pertama yang baik` -**Projek Asal**: [9router by decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Apabila membuka isu, sila jalankan arahan maklumat sistem dan lampirkan fail yang dijana:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Ini menghasilkan `system-info.txt` dengan versi Node.js anda, versi OmniRoute, butiran OS, alat CLI yang dipasang (qoder, gemini, claude, codex, antigravity, droid, dll.), status Docker/PM2 dan pakej sistem — semua yang kami perlukan untuk mengeluarkan semula isu anda dengan cepat. Lampirkan fail terus pada isu GitHub anda.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Setiap pembangun yang menggunakan alatan AI menghadapi masalah ini setiap hari.**OmniRoute dibina untuk menyelesaikan kesemuanya — daripada lebihan kos kepada blok serantau, daripada aliran OAuth yang rosak kepada operasi protokol dan kebolehmerhatian perusahaan. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. - -💸 1. "Saya membayar untuk langganan yang mahal tetapi masih terganggu oleh had" +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Pembangun membayar $20–200/bulan untuk Claude Pro, Codex Pro atau GitHub Copilot. Walaupun membayar, kuota mempunyai siling — 5j penggunaan, had mingguan atau had kadar seminit. Sesi pertengahan pengekodan, pembekal berhenti bertindak balas dan pembangun kehilangan aliran dan produktiviti. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Cara OmniRoute menyelesaikannya:** +**How OmniRoute solves it:** --**Smart 4-Tier Fallback**— Jika kuota langganan habis, diubah hala secara automatik ke API Key → Murah → Percuma tanpa campur tangan manual --**Penjejakan Had Pembekal**— Syot kilat kuota cache dimuat semula pada jadual sebelah pelayan (lalai `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) dengan muat semula manual tersedia dalam UI --**Sokongan Berbilang Akaun**— Berbilang akaun bagi setiap pembekal dengan auto round-robin — apabila satu kehabisan, beralih kepada yang seterusnya --**Kombo Tersuai**— Rantaian sandaran boleh disesuaikan dengan 9 strategi pengimbangan (keutamaan, wajaran, isikan dahulu, pusingan bulat, P2C, rawak, paling kurang digunakan, dioptimumkan kos, rawak ketat) --**Kuota Perniagaan Codex**— Pemantauan kuota ruang kerja Perniagaan/Pasukan terus dalam papan pemuka
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard - -🔌 2. "Saya perlu menggunakan berbilang penyedia tetapi setiap satu mempunyai API yang berbeza" + -OpenAI menggunakan satu format, Claude (Anthropic) menggunakan satu lagi, Gemini satu lagi. Jika pembangun ingin menguji model daripada pembekal yang berbeza atau sandaran antara mereka, mereka perlu mengkonfigurasi semula SDK, menukar titik akhir, menangani format yang tidak serasi. Pembekal tersuai (FriendLI, NIM) mempunyai titik akhir model bukan standard. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Cara OmniRoute menyelesaikannya:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Titik Akhir Disatukan**— Satu `http://localhost:20128/v1` berfungsi sebagai proksi untuk kesemua 60+ pembekal --**Format Terjemahan**— Automatik dan telus: OpenAI ↔ Claude ↔ Gemini ↔ Responses API --**Pembersihan Tindak Balas**— Menghapuskan medan bukan standard (`x_groq`, `penggunaan_pecahan`, `peringkat_perkhidmatan`) yang memecahkan OpenAI SDK v1.83+ --**Penormalan Peranan**— Menukar `pembangun` → `sistem` untuk penyedia bukan OpenAI; `sistem` → `pengguna` untuk GLM/ERNIE --**Think Tag Extraction**— Mengekstrak `` blok daripada model seperti DeepSeek R1 ke dalam `reasoning_content` piawai --**Output Berstruktur untuk Gemini**— `json_schema` → `responseMimeType`/`responseSchema` penukaran automatik --**`strim` lalai kepada `false`**— Menjajarkan dengan spesifikasi OpenAI, mengelakkan SSE yang tidak dijangka dalam Python/Rust/Go SDK
+**How OmniRoute solves it:** - -🌐 3. "Pembekal AI saya menyekat wilayah/negara saya" +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Penyedia seperti OpenAI/Codex menyekat akses daripada kawasan geografi tertentu. Pengguna mendapat ralat seperti `wilayah_negara_negara_tidak disokong` semasa sambungan OAuth dan API. Ini amat mengecewakan bagi pemaju dari negara membangun. + -**Cara OmniRoute menyelesaikannya:** +
+🌐 3. "My AI provider blocks my region/country" --**Konfigurasi Proksi 3 Tahap**— Proksi boleh dikonfigurasikan pada 3 peringkat: global (semua trafik), setiap pembekal (satu pembekal sahaja) dan setiap sambungan/kunci --**Lencana Proksi Berkod Warna**— Penunjuk visual: 🟢 proksi global, 🟡 proksi pembekal, 🔵 proksi sambungan, sentiasa menunjukkan IP --**Pertukaran Token OAuth Melalui Proksi**— Aliran OAuth juga melalui proksi, menyelesaikan `wilayah_wilayah_negara_yang tidak disokong` --**Ujian Sambungan melalui Proksi**— Ujian sambungan menggunakan proksi yang dikonfigurasikan (tiada lagi pintasan langsung) --**Sokongan SOCKS5**— Sokongan proksi SOCKS5 penuh untuk penghalaan keluar --**TLS Fingerprint Spoofing**— Cap jari TLS seperti pelayar melalui `wreq-js` untuk memintas pengesanan bot --**🔏 Padanan Cap Jari CLI**— Menyusun semula pengepala dan medan badan agar sepadan dengan tandatangan binari CLI asli, secara drastik mengurangkan risiko pembenderaan akaun. IP proksi dikekalkan — anda mendapat kedua-dua penyamaran**dan**IP secara serentak
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. - -🆓 4. "Saya mahu menggunakan AI untuk pengekodan tetapi saya tidak mempunyai wang" +**How OmniRoute solves it:** -Tidak semua orang boleh membayar $20–200/bulan untuk langganan AI. Pelajar, pembangun dari negara baru muncul, penggemar dan pekerja bebas memerlukan akses kepada model berkualiti pada kos sifar. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Cara OmniRoute menyelesaikannya:** + --**Pembekal Peringkat Percuma Terbina dalam**— Sokongan asli untuk 100% penyedia percuma: Qoder (5 model tanpa had melalui OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 model tanpa had: qwender3-wenlash3-coplus qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID secara percuma), Gemini CLI (180K token/bulan percuma) --**Ollama Cloud**— Model Ollama dihoskan awan di `api.ollama.com` dengan peringkat "Penggunaan ringan" percuma; gunakan awalan `ollamacloud/` --**Kombo Percuma-Sahaja**— Rantaian `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/bulan dengan masa henti sifar --**Akses Percuma NVIDIA NIM**— ~40 RPM dev-forever akses percuma kepada 70+ model di build.nvidia.com (peralihan daripada kredit kepada had kadar tulen) --**Strategi Dioptimumkan Kos**— Strategi penghalaan yang secara automatik memilih pembekal yang tersedia paling murah +
+🆓 4. "I want to use AI for coding but I have no money" - -🔒 5. "Saya perlu melindungi get laluan AI saya daripada akses tanpa kebenaran" +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -Apabila mendedahkan get laluan AI kepada rangkaian (LAN, VPS, Docker), sesiapa sahaja yang mempunyai alamat boleh menggunakan token/kuota pembangun. Tanpa perlindungan, API terdedah kepada penyalahgunaan, suntikan segera dan penyalahgunaan. +**How OmniRoute solves it:** -**Cara OmniRoute menyelesaikannya:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**Pengurusan Kunci API**— Penjanaan, penggiliran dan skop setiap pembekal dengan halaman `/papan pemuka/pengurus-api` khusus --**Kebenaran Tahap Model**— Hadkan kunci API kepada model tertentu (`openai/*`, corak kad bebas), dengan Togol Benarkan Semua/Sekat --**Perlindungan Titik Akhir API**— Memerlukan kunci untuk `/v1/model` dan sekat pembekal tertentu daripada penyenaraian --**Auth Guard + CSRF Protection**— Semua laluan papan pemuka dilindungi dengan perisian tengah `withAuth` + token CSRF --**Penghad Kadar**— Pengehadan kadar Per-IP dengan tetingkap boleh dikonfigurasikan --**Penapisan IP**— Senarai Benar/senarai sekat untuk kawalan akses --**Pengawal Suntikan Segera**— Pensanitasi terhadap corak segera yang berniat jahat --**Penyulitan AES-256-GCM**— Bukti kelayakan disulitkan semasa rehat
+ - -🛑 6. "Pembekal saya gagal dan saya kehilangan aliran pengekodan saya" +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -Pembekal AI boleh menjadi tidak stabil, mengembalikan ralat 5xx atau mencapai had kadar sementara. Jika pembangun bergantung pada penyedia tunggal, mereka akan terganggu. Tanpa pemutus litar, percubaan semula berulang boleh ranap aplikasi. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Cara OmniRoute menyelesaikannya:** +**How OmniRoute solves it:** --**Pemutus Litar setiap model**— Auto buka/tutup dengan ambang boleh dikonfigurasikan dan cooldown (Tertutup/Buka/Separuh Terbuka), berskop setiap model untuk mengelakkan blok berlatarkan --**Penyingkiran Eksponen**— Kelewatan percubaan semula progresif --**Kawanan Anti Gemuruh**— Mutex + perlindungan semafor terhadap ribut percubaan semula serentak --**Kombo Rantai Sandar**— Jika pembekal utama gagal, secara automatik jatuh melalui rantaian tanpa campur tangan --**Pemutus Litar Kombo**— Lumpuhkan automatik pembekal yang gagal dalam rantaian kombo --**Papan Pemuka Kesihatan**— Pemantauan masa aktif, keadaan pemutus litar, penguncian, statistik cache, kependaman p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest - -🔧 7. "Mengkonfigurasi setiap alat AI adalah membosankan dan berulang" + -Pembangun menggunakan Kursor, Kod Claude, Codex CLI, OpenClaw, Gemini CLI, Kod Kilo... Setiap alat memerlukan konfigurasi yang berbeza (titik akhir API, kunci, model). Mengkonfigurasi semula apabila menukar pembekal atau model adalah membuang masa. +
+🛑 6. "My provider went down and I lost my coding flow" -**Cara OmniRoute menyelesaikannya:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**Papan Pemuka Alat CLI**— Halaman khusus dengan persediaan satu klik untuk Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline --**GitHub Copilot Config Generator**— Menjana `chatLanguageModels.json` untuk Kod VS dengan pemilihan model pukal --**Onboarding Wizard**— Persediaan 4 langkah berpandu untuk pengguna kali pertama --**Satu titik akhir, semua model**— Konfigurasikan `http://localhost:20128/v1` sekali, akses 60+ pembekal
+**How OmniRoute solves it:** - -🔑 8. "Mengurus token OAuth daripada berbilang penyedia adalah neraka" +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Kod Claude, Codex, Gemini CLI, Copilot — semuanya menggunakan OAuth 2.0 dengan token tamat tempoh. Pembangun perlu sentiasa mengesahkan semula, menangani `client_secret is missing`, `redirect_uri_mismatch` dan kegagalan pada pelayan jauh. OAuth pada LAN/VPS amat bermasalah. + -**Cara OmniRoute menyelesaikannya:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Muat Semula Token Auto**— Token OAuth dimuat semula di latar belakang sebelum tamat tempoh --**OAuth 2.0 (PKCE) Terbina dalam**— Aliran automatik untuk Kod Claude, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder --**OAuth Berbilang Akaun**— Berbilang akaun bagi setiap pembekal melalui pengekstrakan token JWT/ID --**OAuth LAN/Remote Fix**— Pengesanan IP peribadi untuk `redirect_uri` + mod URL manual untuk pelayan jauh --**OAuth Behind Nginx**— Menggunakan `window.location.origin` untuk keserasian proksi terbalik --**Panduan OAuth Jauh**— Panduan langkah demi langkah untuk kelayakan Google Cloud pada VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. - -📊 9. "Saya tidak tahu berapa banyak yang saya belanjakan atau di mana" +**How OmniRoute solves it:** -Pembangun menggunakan berbilang penyedia berbayar tetapi tidak mempunyai pandangan bersatu tentang perbelanjaan. Setiap pembekal mempunyai papan pemuka pengebilan sendiri, tetapi tiada paparan disatukan. Kos yang tidak dijangka boleh bertimbun. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Cara OmniRoute menyelesaikannya:** + --**Papan Pemuka Analitis Kos**— Penjejakan kos per-token dan pengurusan belanjawan bagi setiap pembekal --**Had Belanjawan setiap Peringkat**— Siling perbelanjaan setiap peringkat yang mencetuskan sandaran automatik --**Konfigurasi Harga Per-Model**— Harga boleh dikonfigurasikan bagi setiap model --**Statistik Penggunaan Setiap Kunci API**— Kiraan permintaan dan cap masa yang terakhir digunakan setiap kunci --**Papan Pemuka Analitik**— Kad statistik, carta penggunaan model, jadual pembekal dengan kadar kejayaan dan kependaman +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" - -🐛 10. "Saya tidak dapat mendiagnosis ralat dan masalah dalam panggilan AI" +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Apabila panggilan gagal, pembangun tidak tahu sama ada ia adalah had kadar, token tamat tempoh, format yang salah atau ralat pembekal. Log berpecah-belah merentasi terminal yang berbeza. Tanpa pemerhatian, penyahpepijatan adalah percubaan-dan-ralat. +**How OmniRoute solves it:** -**Cara OmniRoute menyelesaikannya:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Papan Pemuka Log Bersatu**— 4 tab: Log Permintaan, Log Proksi, Log Audit, Konsol --**Pemapar Log Konsol**— Pemapar gaya terminal masa nyata dengan tahap berkod warna, tatal automatik, carian, penapis --**Log Proksi SQLite**— Log berterusan yang bertahan dimulakan semula --**Taman Permainan Penterjemah**— 4 mod nyahpepijat: Taman Permainan (terjemahan format), Penguji Sembang (perjalanan pergi balik), Bangku Ujian (batch), Monitor Langsung (masa nyata) --**Permintaan Telemetri**— kependaman p50/p95/p99 + pengesanan X-Request-Id --**Pengelogan Berasaskan Fail dengan Putaran**— Log apl berputar mengikut saiz, hari pengekalan dan kiraan arkib; artifak log panggilan berputar mengikut hari pengekalan dan kiraan fail --**Laporan Maklumat Sistem**— `info sistem jalankan npm` menjana `info-sistem.txt` dengan persekitaran penuh anda (versi Nod, versi OmniRoute, OS, alatan CLI, status Docker/PM2). Lampirkannya apabila melaporkan isu untuk percubaan segera.
+ - -🏗️ 11. "Menyediakan dan menyelenggara pintu masuk adalah rumit" +
+📊 9. "I don't know how much I'm spending or where" -Memasang, mengkonfigurasi dan menyelenggara proksi AI merentas persekitaran yang berbeza (tempatan, VPS, Docker, awan) adalah intensif buruh. Masalah seperti laluan berkod keras, `EACCES` pada direktori, konflik port dan binaan merentas platform menambahkan geseran. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Cara OmniRoute menyelesaikannya:** +**How OmniRoute solves it:** --**npm global install**— `npm install -g omniroute && omniroute` — selesai --**Docker Multi-Platform**— AMD64 + ARM64 asli (Apple Silicon, AWS Graviton, Raspberry Pi) --**Profil Karang Docker**— `asas` (tiada alat CLI) dan `cli` (dengan Kod Claude, Codex, OpenClaw) --**Apl Desktop Elektron**— Apl asli untuk Windows/macOS/Linux dengan dulang sistem, auto mula, mod luar talian --**Mod Split-Port**— API dan Papan Pemuka pada port berasingan untuk senario lanjutan (proksi terbalik, rangkaian kontena) --**Cloud Sync**— Konfigurasikan penyegerakan merentas peranti melalui Cloudflare Workers --**DB Sandaran**— Sandaran automatik, pulihkan, eksport dan import semua tetapan, dengan `DISABLE_SQLITE_AUTO_BACKUP` untuk sandaran yang diuruskan secara luaran
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency - -🌍 12. "Antara muka adalah bahasa Inggeris sahaja dan pasukan saya tidak bercakap bahasa Inggeris" + -Pasukan di negara bukan berbahasa Inggeris, terutamanya di Amerika Latin, Asia dan Eropah, bergelut dengan antara muka bahasa Inggeris sahaja. Halangan bahasa mengurangkan penggunaan dan meningkatkan ralat konfigurasi. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Cara OmniRoute menyelesaikannya:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Papan pemuka i18n — 30 Bahasa**— Semua 500+ kunci diterjemahkan termasuk bahasa Arab, Bulgaria, Denmark, Jerman, Sepanyol, Finland, Perancis, Ibrani, Hindi, Hungary, Indonesia, Itali, Jepun, Korea, Melayu, Belanda, Norway, Poland, Portugis (PT/BR), Romania, Rusia, Slovak, Sweden, Thai, Ukraine, Vietnam, Cina --**Sokongan RTL**— Sokongan kanan ke kiri untuk bahasa Arab dan Ibrani --**README Berbilang Bahasa**— 30 terjemahan dokumentasi lengkap --**Pemilih Bahasa**— Ikon Glob dalam pengepala untuk penukaran masa nyata
+**How OmniRoute solves it:** - -🔄 13. "Saya perlukan lebih daripada sembang — saya perlukan pembenaman, imej, audio" +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -AI bukan sekadar penyelesaian sembang. Pembangun perlu menjana imej, menyalin audio, membuat pembenaman untuk RAG, menyusun semula dokumen dan kandungan sederhana. Setiap API mempunyai titik akhir dan format yang berbeza. + -**Cara OmniRoute menyelesaikannya:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Pembenaman**— `/v1/pembenaman` dengan 6 pembekal dan 9+ model --**Penjanaan Imej**— `/v1/imej/generasi` dengan 10 pembekal dan 20+ model (OpenAI, xAI, Bersama-sama, Bunga Api, Nebius, Hiperbolik, NanoBanana, Antigraviti, SD WebUI, ComfyUI) --**Teks-ke-Video**— `/v1/video/generasi` — ComfyUI (AnimateDiff, SVD) dan SD WebUI --**Teks-ke-Muzik**— `/v1/muzik/generasi` — ComfyUI (Audio Terbuka, MusicGen) --**Transkripsi Audio**— `/v1/audio/transkripsi` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Text-to-Speech**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + pembekal sedia ada --**Moderations**— `/v1/moderations` — Pemeriksaan keselamatan kandungan --**Penyediaan semula**— `/v1/rerank` — Penyusunan semula perkaitan dokumen --**Responses API**— Sokongan `/v1/respons` penuh untuk Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. - -🧪 14. "Saya tiada cara untuk menguji dan membandingkan kualiti merentas model" +**How OmniRoute solves it:** -Pembangun ingin mengetahui model mana yang terbaik untuk kes penggunaan mereka — kod, terjemahan, penaakulan — tetapi membandingkan secara manual adalah perlahan. Tiada alat eval bersepadu wujud. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**Cara OmniRoute menyelesaikannya:** + --**LLM Evaluations**— Ujian set emas dengan 10 kes pra-muat meliputi salam, matematik, geografi, penjanaan kod, pematuhan JSON, terjemahan, penurunan harga, penolakan keselamatan --**4 Strategi Padanan**— `tepat`, `mengandungi`, `regex`, `custom` (fungsi JS) --**Bangku Ujian Taman Permainan Penterjemah**— Ujian kelompok dengan berbilang input dan output yang dijangka, perbandingan merentas pembekal --**Penguji Sembang**— Perjalanan pergi balik penuh dengan pemaparan respons visual --**Pantau Langsung**— Strim masa nyata semua permintaan yang mengalir melalui proksi +
+🌍 12. "The interface is English-only and my team doesn't speak English" - -📈 15. "Saya perlu membuat skala tanpa kehilangan prestasi" +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -Apabila volum permintaan bertambah, tanpa menyimpan cache soalan yang sama menjana kos pendua. Tanpa idempotensi, pendua meminta pemprosesan sisa. Had kadar setiap pembekal mesti dipatuhi. +**How OmniRoute solves it:** -**Cara OmniRoute menyelesaikannya:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Cache Semantik**— Cache dua peringkat (tandatangan + semantik) mengurangkan kos dan kependaman --**Request Idempotency**— tetingkap penyahduplikasi 5s untuk permintaan yang sama --**Pengesanan Had Kadar**— RPM setiap pembekal, jurang min dan penjejakan serentak maks --**Had Kadar Boleh Diedit**— Lalai boleh dikonfigurasikan dalam Tetapan → Ketahanan dengan kegigihan --**Cache Pengesahan Kunci API**— Cache 3 peringkat untuk prestasi pengeluaran --**Papan Pemuka Kesihatan dengan Telemetri**— kependaman p50/p95/p99, statistik cache, masa beroperasi
+ - -🤖 16. "Saya mahu mengawal tingkah laku model secara global" +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Pembangun yang mahukan semua respons dalam bahasa tertentu, dengan nada tertentu atau ingin mengehadkan token penaakulan. Mengkonfigurasi ini dalam setiap alat/permintaan adalah tidak praktikal. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Cara OmniRoute menyelesaikannya:** +**How OmniRoute solves it:** --**System Prompt Injection**— Gesaan global digunakan untuk semua permintaan --**Pengesahan Belanjawan Berfikir**— Kawalan peruntukan token penaakulan setiap permintaan (laluan, auto, tersuai, adaptif) --**9 Strategi Penghalaan**— Strategi global yang menentukan cara permintaan diedarkan --**Penghala Wildcard**— corak `penyedia/*` menghala secara dinamik ke mana-mana pembekal --**Kombo Dayakan/Lumpuhkan Togol**— Togol kombo terus dari papan pemuka --**Togol Pembekal**— Dayakan/lumpuhkan semua sambungan untuk pembekal dengan satu klik --**Pembekal Disekat**— Kecualikan pembekal khusus daripada penyenaraian `/v1/models`
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex - -🧰 17. "Saya memerlukan alatan MCP sebagai keupayaan produk kelas pertama" + -Banyak get laluan AI mendedahkan MCP hanya sebagai butiran pelaksanaan tersembunyi. Pasukan memerlukan lapisan operasi yang boleh dilihat dan boleh diurus. +
+🧪 14. "I have no way to test and compare quality across models" -**Cara OmniRoute menyelesaikannya:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP muncul dalam navigasi papan pemuka dan tab protokol titik akhir -- Halaman pengurusan MCP khusus dengan proses, alatan, skop dan audit -- Permulaan pantas terbina dalam untuk `omniroute --mcp` dan onboarding klien
+**How OmniRoute solves it:** - -🧠 18. "Saya memerlukan orkestrasi A2A dengan laluan tugas penyegerakan + strim" +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Aliran kerja ejen memerlukan balasan langsung dan pelaksanaan strim jangka panjang dengan kawalan kitaran hayat. + -**Cara OmniRoute menyelesaikannya:** +
+📈 15. "I need to scale without losing performance" -- Titik akhir JSON-RPC A2A (`POST /a2a`) dengan `mesej/hantar` dan `message/strim` -- Penstriman SSE dengan penyebaran keadaan terminal -- API kitaran hayat tugas untuk `tugas/dapat` dan `tugas/batal`
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. - -🛰️ 19. "Saya memerlukan kesihatan proses MCP sebenar, bukan status yang diduga" +**How OmniRoute solves it:** -Pasukan operasi perlu mengetahui sama ada MCP sebenarnya masih hidup, bukan hanya sama ada API boleh dicapai. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**Cara OmniRoute menyelesaikannya:** + -- Fail degupan jantung masa jalan dengan PID, cap masa, pengangkutan, kiraan alat dan mod skop -- API status MCP yang menggabungkan degupan jantung + aktiviti terkini -- Kad status UI untuk kesegaran proses/masa hidup/degupan jantung +
+🤖 16. "I want to control model behavior globally" - -📋 20. "Saya memerlukan pelaksanaan alat MCP yang boleh diaudit" +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Apabila alat mengubah konfigurasi atau mencetuskan tindakan ops, pasukan memerlukan kebolehkesanan forensik. +**How OmniRoute solves it:** -**Cara OmniRoute menyelesaikannya:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- Pengelogan audit yang disokong SQLite untuk panggilan alat MCP -- Tapis mengikut alat, kejayaan/kegagalan, kunci API dan penomboran -- Jadual audit papan pemuka + titik akhir statistik untuk automasi
+ - -🔐 21. "Saya memerlukan keizinan MCP berskop bagi setiap penyepaduan" +
+🧰 17. "I need MCP tools as first-class product capabilities" -Pelanggan yang berbeza harus mempunyai akses paling tidak istimewa kepada kategori alat. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Cara OmniRoute menyelesaikannya:** +**How OmniRoute solves it:** -- 10 skop MCP berbutir untuk akses alat terkawal -- Penguatkuasaan skop dan keterlihatan dalam UI pengurusan MCP -- Postur lalai yang selamat untuk perkakas operasi
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding - -⚙️ 22. "Saya memerlukan kawalan operasi tanpa mengatur semula" + -Pasukan memerlukan perubahan masa jalan yang cepat semasa insiden atau peristiwa kos. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**Cara OmniRoute menyelesaikannya:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Tukar pengaktifan kombo terus dari papan pemuka MCP -- Gunakan profil daya tahan daripada pek dasar yang telah ditetapkan -- Tetapkan semula keadaan pemutus litar daripada panel operasi yang sama
+**How OmniRoute solves it:** - -🔄 23. "Saya memerlukan keterlihatan dan pembatalan kitaran hayat tugas A2A secara langsung" +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Tanpa keterlihatan kitaran hayat, insiden tugasan menjadi sukar untuk dicuba. + -**Cara OmniRoute menyelesaikannya:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Penyenaraian tugas/penapisan mengikut keadaan/kemahiran dengan penomboran -- Latih tubi tentang metadata tugas, peristiwa dan artifak -- Titik akhir pembatalan tugas dan tindakan UI dengan pengesahan
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. - -🌊 24. "Saya memerlukan metrik strim aktif untuk beban A2A" +**How OmniRoute solves it:** -Aliran kerja penstriman memerlukan cerapan operasi tentang konkurensi dan sambungan langsung. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**Cara OmniRoute menyelesaikannya:** + -- Kaunter aliran aktif disepadukan ke dalam status A2A -- Cap masa tugas terakhir dan kiraan setiap negeri -- Kad papan pemuka A2A untuk pemantauan operasi masa nyata +
+📋 20. "I need auditable MCP tool execution" - -🪪 25. "Saya memerlukan penemuan ejen standard untuk pelanggan" +When tools mutate config or trigger ops actions, teams need forensic traceability. -Pelanggan dan orkestra luar memerlukan metadata yang boleh dibaca mesin untuk onboarding. +**How OmniRoute solves it:** -**Cara OmniRoute menyelesaikannya:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- Kad Ejen terdedah di `/.well-known/agent.json` -- Keupayaan dan kemahiran ditunjukkan dalam UI pengurusan -- API status A2A termasuk metadata penemuan untuk automasi
+ - -🧭 26. "Saya memerlukan kebolehtemuan protokol dalam UX produk" +
+🔐 21. "I need scoped MCP permissions per integration" -Jika pengguna tidak dapat menemui permukaan protokol, penggunaan dan kualiti sokongan akan menurun. +Different clients should have least-privilege access to tool categories. -**Cara OmniRoute menyelesaikannya:** +**How OmniRoute solves it:** -- Halaman**Titik Akhir**yang disatukan dengan tab untuk Titik Akhir Proksi, MCP, A2A dan API -- Togol status perkhidmatan dalam talian (Dalam Talian/Luar Talian) untuk MCP dan A2A -- Pautan dari gambaran keseluruhan ke tab pengurusan khusus
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling - -🧪 27. "Saya memerlukan pengesahan protokol hujung ke hujung dengan pelanggan sebenar" + -Ujian olok-olok tidak mencukupi untuk mengesahkan keserasian protokol sebelum dikeluarkan. +
+⚙️ 22. "I need operational controls without redeploying" -**Cara OmniRoute menyelesaikannya:** +Teams need quick runtime changes during incidents or cost events. -- Suite E2E yang but apl dan menggunakan pengangkutan pelanggan MCP SDK sebenar -- Ujian pelanggan A2A untuk penemuan, menghantar, menstrim, mendapatkan dan membatalkan aliran -- Periksa silang dakwaan terhadap audit MCP dan API tugasan A2A
+**How OmniRoute solves it:** - -📡 28. "Saya memerlukan kebolehmerhatian bersatu merentas semua antara muka" +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -Memisahkan kebolehmerhatian mengikut protokol mewujudkan titik buta dan MTTR yang lebih panjang. + -**Cara OmniRoute menyelesaikannya:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Papan pemuka/log/analisis bersatu dalam satu produk -- Kesihatan + audit + telemetri permintaan merentas lapisan OpenAI, MCP dan A2A -- API Operasi untuk status dan automasi
+Without lifecycle visibility, task incidents become hard to triage. - -💼 29. "Saya memerlukan satu masa jalan untuk proksi + alatan + orkestrasi ejen" +**How OmniRoute solves it:** -Menjalankan banyak perkhidmatan berasingan meningkatkan kos operasi dan mod kegagalan. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**Cara OmniRoute menyelesaikannya:** + -- Proksi serasi OpenAI, pelayan MCP dan pelayan A2A dalam satu tindanan -- Kebenaran dikongsi, daya tahan, stor data dan kebolehmerhatian -- Model dasar yang konsisten merentas semua permukaan interaksi +
+🌊 24. "I need active stream metrics for A2A load" - -🚀 30. "Saya perlu menghantar aliran kerja agen tanpa rebakan kod gam" +Streaming workflows require operational insight into concurrency and live connections. -Pasukan kehilangan halaju apabila mencantumkan berbilang perkhidmatan dan skrip ad-hoc. +**How OmniRoute solves it:** -**Cara OmniRoute menyelesaikannya:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Strategi titik akhir bersatu untuk pelanggan dan ejen -- UI pengurusan protokol terbina dalam dan laluan pengesahan asap -- Asas sedia pengeluaran (keselamatan, pembalakan, daya tahan, sandaran)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Playbook A: Maksimumkan langganan berbayar + sandaran murah**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -607,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Playbook B: Timbunan pengekodan kos sifar**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Playbook C: 24/7 rantai sandaran sentiasa hidup**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -630,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Playbook D: Operasi ejen dengan MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Sediakan pengekodan AI dalam beberapa minit pada**$0/bulan**. Sambungkan akaun percuma ini dan gunakan kombo**Timbunan Percuma**terbina dalam. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Langkah | Tindakan | Pembekal Dibuka Kunci | -| ---- | -------------------------------------------------- | ----------------------------------------------------------------- | -| 1 | Sambung**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**tanpa had**| -| 2 | Sambung**Qoder**(Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... —**tanpa had**| -| 3 | Sambung**Qwen**(Kod Peranti) | qwen3-coder-plus, qwen3-coder-flash... —**tanpa had**| -| 4 | Sambung**Gemini CLI**(Google OAuth) | gemini-3-flash, Gemini-2.5-pro —**180K/bln percuma**| -| 5 | `/papan pemuka/kombo` →**Templat Timbunan Percuma ($0)**| Round-robin semua pembekal percuma secara automatik | +| Step | Action | Providers Unlocked | +| ---- | -------------------------------------------------- | ------------------------------------------------------------------ | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Tuding mana-mana IDE/CLI ke:**`http://localhost:20128/v1` · Kunci API: `any-string` · Selesai. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Liputan tambahan pilihan (juga percuma):**Kunci API Groq (30 RPM percuma), NVIDIA NIM (40 RPM percuma, 70+ model), Cerebras (1M tok/hari), kunci API LongCat (50M token/hari!), Cloudflare Workers AI (10K Neuron/hari, 50+ model).## Mula Pantas +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Mula Pantas ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **pengguna pnpm:**Jalankan `pnpm approve-builds -g` selepas pemasangan untuk mendayakan skrip binaan asli yang diperlukan oleh `better-sqlite3` dan `@swc/core`: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > > ```bash > pnpm install -g omniroute -> pnpm approve-builds -g # Pilih semua pakej → approve +> pnpm approve-builds -g # Select all packages → approve > omniroute > ``` -Papan pemuka dibuka di `http://localhost:20128` dan URL asas API ialah `http://localhost:20128/v1`. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Perintah | Penerangan | -| ----------------------------- | ------------------------------------------------------------------------ | -| `laluan omni` | Mulakan pelayan (`PORT=20128`, API dan papan pemuka pada port yang sama) | -| `laluan omni --port 3000` | Tetapkan port kanonik/API kepada 3000 | -| `laluan omni --mcp` | Mulakan pelayan MCP (stdio transport) | -| `laluan omni --tidak-terbuka` | Jangan auto buka penyemak imbas | -| `omniroute --help` | Tunjukkan bantuan | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Mod split-port pilihan:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -Untuk kebanyakan penempatan, anda hanya perlu: +For most deployments, you only need: -| Pembolehubah | Lalai | Tujuan | -| ------------------------- | ---------------------------- | ----------------------------------------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600000` | Garis dasar dikongsi untuk pengambilan huluan, tamat masa Undici tersembunyi, permintaan cap jari TLS dan permintaan jambatan API/masa tamat proksi | -| `STREAM_IDLE_TIMEOUT_MS` | mewarisi `REQUEST_TIMEOUT_MS` | Jurang maksimum antara bahagian penstriman sebelum OmniRoute membatalkan strim SSE | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -Keserasian ke belakang dikekalkan: `FETCH_TIMEOUT_MS` sedia ada, `API_BRIDGE_PROXY_TIMEOUT_MS` dan var tamat masa setiap lapisan lain masih berfungsi dan mengatasi garis dasar yang dikongsi. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Penggantian lanjutan tersedia jika anda memerlukan kawalan yang lebih halus:| Pembolehubah | Lalai | Tujuan | -| --------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | mewarisi `REQUEST_TIMEOUT_MS` | Jumlah tamat masa permintaan huluan yang digunakan oleh isyarat pengguguran pengambilan utama | -| `FETCH_HEADERS_TIMEOUT_MS` | mewarisi `FETCH_TIMEOUT_MS` | Had masa Undici untuk menerima pengepala respons huluan | -| `FETCH_BODY_TIMEOUT_MS` | mewarisi `FETCH_TIMEOUT_MS` | Had masa Undici antara ketulan badan hulu (`0` melumpuhkannya) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Tamat masa sambungan Undici TCP | -| `FETCH_KEEPLIVE_TIMEOUT_MS` | `4000` | Undici melahu keep-alive soket tamat masa | -| `TLS_CLIENT_TIMEOUT_MS` | mewarisi `FETCH_TIMEOUT_MS` | Tamat masa untuk permintaan cap jari TLS yang dibuat melalui `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | mewarisi `REQUEST_TIMEOUT_MS` atau `30000` | Tamat masa untuk pemajuan proksi `/v1` daripada port API ke port papan pemuka | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `maks(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Tamat masa permintaan masuk pada pelayan jambatan API | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Tamat masa pengepala masuk pada pelayan jambatan API | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Simpan-hidup tamat masa pada pelayan jambatan API | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Tamat masa ketidakaktifan soket pada pelayan jambatan API (`0` melumpuhkannya) | +Advanced overrides are available if you need finer control: -Jika anda menjalankan OmniRoute di belakang Nginx, Caddy, Cloudflare atau proksi terbalik yang lain, pastikan proksi -tamat masa juga lebih tinggi daripada tamat masa strim OmniRoute/ambil anda.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Buka Papan Pemuka → `Penyedia` dan sambungkan sekurang-kurangnya satu pembekal (kunci OAuth atau API). -2. Buka Papan Pemuka → `Titik Akhir` dan buat kunci API. -3. (Pilihan) Buka Papan Pemuka → `Kombo` dan tetapkan rantai sandaran anda.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Berfungsi dengan Kod Claude, Codex CLI, Gemini CLI, Kursor, Cline, OpenClaw, OpenCode dan SDK yang serasi dengan OpenAI.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (untuk operasi dipacu alat):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -Kemudian sambungkan klien MCP anda melalui `stdio` dan alat ujian seperti: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` -- `kombo_senarai_omniroute` +- `omniroute_list_combos` -**A2A (untuk aliran kerja ejen-ke-ejen):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -759,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Suite ini mengesahkan aliran klien MCP dan A2A sebenar terhadap apl yang sedang berjalan.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -767,13 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` - -Void Linux (templat `xbps-src`) +
+Void Linux (`xbps-src` template) -Untuk pengguna Void Linux, anda boleh membina pakej asli menggunakan `xbps-src`. Simpan blok ini sebagai `srcpkgs/omniroute/template`:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -785,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -793,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -869,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -880,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute tersedia sebagai imej Docker awam di [Hab ​​Docker](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Larian pantas:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -890,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**Dengan fail persekitaran:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Menggunakan Docker Compose:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Sokongan papan pemuka untuk penempatan Docker kini termasuk**Terowong Pantas Cloudflare**satu klik pada `Papan Pemuka → Titik Akhir`. Yang pertama membolehkan muat turun `cloudflared` hanya apabila diperlukan, memulakan terowong sementara ke titik akhir `/v1` semasa anda dan menunjukkan URL `https://*.trycloudflare.com/v1` yang dijana terus di bawah URL awam biasa anda. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Nota: +Notes: -- URL Terowong Pantas adalah sementara dan berubah selepas setiap kali dimulakan semula. -- Terowong Pantas tidak dipulihkan secara automatik selepas OmniRoute atau kontena dimulakan semula. Dayakan semula mereka dari papan pemuka apabila diperlukan. -- Pemasangan terurus kini menyokong Linux, macOS dan Windows pada `x64` / `arm64`. -- Terowong Pantas Terurus lalai kepada pengangkutan HTTP/2 untuk mengelakkan amaran penimbal UDP QUIC yang bising dalam persekitaran kontena yang dikekang. Tetapkan `CLOUDFLARED_PROTOCOL=quic` atau `auto` jika anda mahukan pengangkutan yang berbeza. -- Imej docker menggabungkan akar CA sistem dan menyerahkannya kepada `cloudflared` terurus, yang mengelakkan kegagalan kepercayaan TLS apabila tali but terowong berada di dalam bekas. -- SQLite berjalan dalam mod WAL. `docker stop` harus dibenarkan selesai supaya OmniRoute boleh menyemak semula perubahan terkini ke `storage.sqlite`. -- Fail Karang yang digabungkan telah menetapkan tempoh tangguh hentian 40-an. Jika anda menjalankan imej secara terus, pastikan `--stop-timeout 40` (atau serupa) supaya hentian manual tidak memotong pembersihan penutupan. -- Tetapkan `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` jika anda mahu OmniRoute menggunakan binari sedia ada dan bukannya memuat turun satu. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Menggunakan Docker Compose dengan Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute boleh didedahkan dengan selamat menggunakan peruntukan SSL automatik Caddy. Pastikan rekod DNS A domain anda menghala ke IP pelayan anda.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` +| Image | Tag | Size | Description | +| ------------------------ | -------- | ------ | --------------------- | +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | -| Imej | Tag | Saiz | Penerangan | -| ------------------------- | -------- | ------ | ---------------------- | -| `diegosouzapw/omniroute` | `terkini` | ~250MB | Keluaran stabil terkini | -| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Versi semasa |--- +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**BARU!**OmniRoute kini tersedia sebagai**aplikasi desktop asli**untuk Windows, macOS dan Linux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Jalankan OmniRoute sebagai apl desktop kendiri — tiada terminal, tiada penyemak imbas, tiada internet diperlukan untuk model tempatan. Aplikasi berasaskan Elektron termasuk: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Tetingkap Asli**— Tetingkap apl khusus dengan penyepaduan dulang sistem -- 🔄**Auto-Mula**— Lancarkan OmniRoute pada log masuk sistem -- 🔔**Pemberitahuan Asli**— Dapatkan makluman untuk masalah keletihan kuota atau pembekal -- ⚡**Pasang Satu Klik**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Mod Luar Talian**— Berfungsi sepenuhnya di luar talian dengan pelayan yang digabungkan### Mula Pantas +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Mula Pantas ```bash # Development mode @@ -979,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -Apabila diminimumkan, OmniRoute tinggal dalam dulang sistem anda dengan tindakan pantas: +When minimized, OmniRoute lives in your system tray with quick actions: -- Buka papan pemuka -- Tukar port pelayan -- Hentikan permohonan +- Open dashboard +- Change server port +- Quit application -📖 Dokumentasi penuh: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Peringkat | Pembekal | Kos | Set Semula Kuota | Terbaik Untuk | -| ---------------- | --------------------------- | ------------------------------ | ------------------ | ------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 LANGGANAN** | Kod Claude (Pro) | $20/bln | 5j + mingguan | Sudah melanggan | -| | Codex (Plus/Pro) | $20-200/bln | 5j + mingguan | Pengguna OpenAI | -| | Gemini CLI | **PERCUMA** | 180K/bln + 1K/hari | Semua orang! | -| | GitHub Copilot | $10-19/bln | Bulanan | Pengguna GitHub | -| **🔑 KUNCI API** | NVIDIA NIM | **PERCUMA**(dev forever) | ~40 RPM | 70+ model terbuka | -| | Serebral | **PERCUMA**(1J tok/hari) | 60K TPM / 30 RPM | Terpantas di dunia | -| | Groq | **PERCUMA**(30 RPM) | 14.4K RPD | Llama/Gemma sangat pantas | -| | DeepSeek V3.2 | $0.27/$1.10 setiap 1J | Tiada | Penalaran harga/kualiti terbaik | -| | xAI Grok-4 Cepat | **$0.20/$0.50 setiap 1J**🆕 | Tiada | Panggilan alat + terpantas, ultralow | -| | xAI Grok-4 (standard) | $0.20/$1.50 setiap 1J 🆕 | Tiada | Penaakulan perdana daripada xAI | -| | Mistral | Percubaan percuma + berbayar | Kadar terhad | AI Eropah | -| | OpenRouter | Bayar setiap penggunaan | Tiada | 100+ model aggr. | -| **💰 MURAH** | GLM-5 (melalui Z.AI) 🆕 | $0.5/1J | Setiap hari 10AM | Keluaran 128K, perdana terbaharu | -| | GLM-4.7 | $0.6/1J | Setiap hari 10AM | Sandaran belanjawan | -| | MiniMax M2.5 🆕 | Input $0.3/1J | 5 jam bergolek | Penaakulan + tugas agen | -| | MiniMax M2.1 | $0.2/1J | 5 jam bergolek | Pilihan termurah | -| | Kimi K2.5 (Moonshot API) 🆕 | Bayar setiap penggunaan | Tiada | Akses langsung Moonshot API | -| | Kimi K2 | $9/bln flat | 10 juta token/bln | Kos yang boleh diramal | -| **🆓 PERCUMA** | Qoder | **$0** | tanpa had | 5 model tanpa had | -| | Qwen | **$0** | tanpa had | 4 model tanpa had | -| | Kiro | **$0** | tanpa had | Claude Sonnet/Haiku (Pembina AWS) | -| | LongCat Flash-Lite 🆕 | **$0**(50J tok/hari 🔥) | 1 RPS | Kuota percuma terbesar di Bumi | -| | Pendebungaan AI 🆕 | **$0**(tiada kunci diperlukan) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | -| | Pekerja Cloudflare AI 🆕 | **$0**(10K Neuron/hari) | ~150 resp/hari | 50+ model, kelebihan global | -| | Scaleway AI 🆕 | **$0**(Jumlah token 1J) | Kadar terhad | EU/GDPR, Qwen3 235B, Llama 70B | > 🆕**Model baharu ditambahkan (Mac 2026):**Keluarga Grok-4 Fast pada $0.20/$0.50/M (ditanda aras pada 1143ms — 30% lebih pantas daripada Gemini 2.5 Flash), GLM-5 melalui Z.AI dengan output 128K, penaakulan MiniMax M2.5. Kim2 DeepSeek V3. Penaakulan langsung Kim2, DeepSeek V3. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 Timbunan Kombo $0 — Persediaan Percuma Lengkap:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Kos sifar. Jangan sekali-kali berhenti pengekodan.**Konfigurasikan ini sebagai satu kombo OmniRoute dan semua sandaran berlaku secara automatik — tiada penukaran manual.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Semua model di bawah adalah**100% percuma dengan kad kredit sifar diperlukan**. Laluan automatik OmniRoute antara mereka apabila satu kuota habis — gabungkan kesemuanya untuk kombo $0 yang tidak boleh dipecahkan.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Model | Awalan | Had | Had Kadar | -| ------------------- | ------ | ------------- | ---------------------- | -| `claude-sonnet-4.5` | `kr/` |**Tidak terhad**| Tiada had harian dilaporkan | -| `claude-haiku-4.5` | `kr/` |**Tidak terhad**| Tiada had harian dilaporkan | -| `claude-opus-4.6` | `kr/` |**Tidak terhad**| Opus terkini melalui Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) -| Model | Awalan | Had | Had Kadar | -| ------------------- | ------ | ------------- | --------------- | -| `kimi-k2-berfikir` | `jika/` |**Tidak terhad**| Tiada had dilaporkan | -| `qwen3-coder-plus` | `jika/` |**Tidak terhad**| Tiada had dilaporkan | -| `deepseek-r1` | `jika/` |**Tidak terhad**| Tiada had dilaporkan | -| `minimaks-m2.1` | `jika/` |**Tidak terhad**| Tiada had dilaporkan | -| `kimi-k2` | `jika/` |**Tidak terhad**| Tiada had dilaporkan | +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | --------------------- | +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -> Kaedah sambungan yang disyorkan:**Token Akses Peribadi + `qodercli`**. Pelayar OAuth ialah -> percubaan dan dilumpuhkan secara lalai melainkan pembolehubah persekitaran `QODER_OAUTH_*` dikonfigurasikan.### 🟡 QWEN MODELS (Device Code Auth) +### 🟢 QODER MODELS (Free PAT via qodercli) -| Model | Awalan | Had | Had Kadar | +| Model | Prefix | Limit | Rate Limit | +| ------------------ | ------ | ------------- | --------------- | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | + +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. + +### 🟡 QWEN MODELS (Device Code Auth) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | ------------------- | -| `qwen3-coder-plus` | `qw/` |**Tidak terhad**| Tiada had yang dilaporkan | -| `qwen3-coder-flash` | `qw/` |**Tidak terhad**| Tiada had yang dilaporkan | -| `qwen3-coder-next` | `qw/` |**Tidak terhad**| Tiada had yang dilaporkan | -| `model penglihatan` | `qw/` |**Tidak terhad**| Multimodal (imej) |### 🟣 GEMINI CLI (Google OAuth) +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Model | Awalan | Had | Had Kadar | -| ------------------------- | ------ | --------------------------- | ------------- | -| `gemini-3-flash-preview` | `gc/` |**180K tok/bulan**+ 1K/hari | Tetapan semula bulanan | -| `gemini-2.5-pro` | `gc/` | 180K/bulan (kolam kongsi) | Berkualiti tinggi |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +### 🟣 GEMINI CLI (Google OAuth) -| Peringkat | Had Harian | Had Kadar | Nota | -| ---------- | ------------ | ----------- | ------------------------------------------------------------------- | -| Percuma (Dev) | Tiada topi token |**~40 RPM**| 70+ model; beralih kepada had kadar tulen pertengahan 2025 | +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -Model percuma yang popular: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-deepseek`,/`deepseek`r### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) -| Peringkat | Had Harian | Had Kadar | Nota | -| ---- | ----------------- | ---------------- | ------------------------------------------ | -| Percuma |**1J token/hari**| 60K TPM / 30 RPM | Inferens LLM terpantas di dunia; ditetapkan semula setiap hari | +| Tier | Daily Limit | Rate Limit | Notes | +| ---------- | ------------ | ----------- | ------------------------------------------------------ | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -Tersedia secara percuma: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -| Peringkat | Had Harian | Had Kadar | Nota | -| ---- | ------------- | ---------------- | ---------------------------------------- | -| Percuma |**14.4K RPD**| 30 RPM setiap model | Tiada kad kredit; 429 pada had, tidak dicaj | +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -Tersedia secara percuma: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -| Model | Awalan | Kuota Percuma Harian | Nota | -| ---------------------------- | ------ | ----------------- | ------------------------ | -| `LongCat-Flash-Lite` | `lc/` |**50J token**💥 | Kuota percuma terbesar pernah | -| `LongCat-Flash-Chat` | `lc/` | 500K token | Sembang berbilang pusingan | -| `LongCat-Flash-Thinking` | `lc/` | 500K token | Penaakulan / CoT | -| `LongCat-Flash-Thinking-2601` | `lc/` | 500K token | Versi Jan 2026 | -| `LongCat-Flash-Omni-2603` | `lc/` | 500K token | Multimodal | +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -> 100% percuma semasa dalam beta awam. Daftar di [longcat.chat](https://longcat.chat) dengan e-mel atau telefon. Ditetapkan semula setiap hari 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +### 🔴 GROQ (Free API Key — console.groq.com) -| Model | Awalan | Had Kadar | Pembekal Di Belakang | -| ---------- | ------ | ---------- | ------------------- | -| `openai` | `pol/` | 1 req/15s | GPT-5 | -| `claude` | `pol/` | 1 req/15s | Claude Anthropic | -| `gemini` | `pol/` | 1 req/15s | Google Gemini | -| `mendalam` | `pol/` | 1 req/15s | DeepSeek V3 | -| `llama` | `pol/` | 1 req/15s | Pengakap Meta Llama 4 | -| `mistral` | `pol/` | 1 req/15s | Mistral AI | +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ------------- | ---------------- | ----------------------------------------- | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -> ✨**Sifar geseran:**Tiada pendaftaran, tiada kunci API. Tambahkan pembekal Pendebungaan dengan medan kunci kosong dan ia berfungsi serta-merta.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -| Peringkat | Neuron Harian | Penggunaan Setara | Nota | -| ---- | ------------- | ---------------------------------------------------- | ------------------------ | -| Percuma |**10,000**| ~150 LLM respons / 500s audio / 15K benam | Kelebihan global, 50+ model | +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 -Model percuma yang popular: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (audio percuma!), `@cf/qwen/qwen2.5-coder-15b +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | -> Memerlukan Token API + ID Akaun daripada [dash.cloudflare.com](https://dash.cloudflare.com). Simpan ID Akaun dalam tetapan pembekal.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. -| Peringkat | Kuota Percuma | Lokasi | Nota | +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | +| ---------- | ------ | ---------- | ------------------ | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | + +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. + +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 + +| Tier | Daily Neurons | Equivalent Usage | Notes | +| ---- | ------------- | --------------------------------------- | ----------------------- | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | + +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` + +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. + +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | | ---- | ------------- | ------------ | ----------------------------------- | -| Percuma |**1J token**| 🇫🇷 Paris, EU | Tiada kad kredit diperlukan dalam had | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | -Tersedia secara percuma: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` -> patuh EU/GDPR. Dapatkan kunci API di [console.scaleway.com](https://console.scaleway.com). +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). ->**💡 Timbunan Percuma Terunggul (11 Pembekal, $0 Selamanya):** +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -> Qoder (jika/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 juta token/hari 🔥 -> Pendebungaan (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — tiada kunci diperlukan -> Qwen (qw/) → model qwen3-coder TANPA HAD -> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/hari percuma -> Cloudflare AI (cf/) → 50+ model — 10K Neuron/hari -> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M token percuma (EU) -> Groq (groq/) → Llama/Gemma — 14.4K req/hari sangat pantas -> NVIDIA NIM (nvidia/) → 70+ model terbuka — 40 RPM selama-lamanya -> Cerebras (cerebras/) → Llama/Qwen terpantas dunia — 1M tok/hari -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Transkripsikan sebarang audio/video untuk**$0**— Deepgram memimpin dengan percuma $200, sandaran AssemblyAI $50, Groq Whisper sebagai sandaran kecemasan tanpa had. +## 🎙️ Free Transcription Combo -| Pembekal | Kredit Percuma | Model Terbaik | Had Kadar | -| ----------------- | ----------------------- | ------------------------------------------ | ---------------------------- | -| 🟢**Deepgram**|**$200 percuma**(pendaftaran) | `nova-3` — ketepatan terbaik, 30+ bahasa | Tiada had RPM untuk kredit percuma | -| 🔵**PerhimpunanAI**|**$50 percuma**(pendaftaran) | `universal-3-pro` — bab, sentimen, PII | Tiada had RPM untuk kredit percuma | -| 🔴**Groq**|**Percuma selamanya**| `whisper-large-v3` — OpenAI Whisper | 30 RPM (kadar terhad) | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. -**Kombo yang dicadangkan dalam `/papan pemuka/kombo`:**``` +| Provider | Free Credits | Best Model | Rate Limit | +| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | + +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Kemudian dalam tab `/papan pemuka/media` →**Transkripsi**: muat naik mana-mana fail audio atau video → pilih titik akhir kombo anda → dapatkan transkripsi dalam format yang disokong.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 dibina sebagai platform operasi, bukan hanya proksi geganti.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Ciri | Apa yang Dilakukan | -| ---------------------------------------- | ------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Keluarga Cepat Grok-4** | model xAI pada $0.20/$0.50/M — menanda aras 1143ms (30% lebih pantas daripada Gemini 2.5 Flash) | -| 🧠**GLM-5 melalui Z.AI** | Konteks keluaran 128K, $0.5/1J — perdana terbaharu daripada keluarga GLM | -| 🔮**MiniMax M2.5** | Penaakulan + tugas agen pada $0.30/1J — peningkatan ketara daripada M2.1 | -| 🎯**alatMemanggil Bendera setiap Model** | Per-model `toolCalling: true/false` dalam registry — AutoCombo melangkau model bukan alat yang mampu | -| 🌍**Pengesanan Niat Pelbagai Bahasa** | Kata kunci PT/ZH/ES/AR dalam pemarkahan AutoCombo — pemilihan model yang lebih baik untuk kandungan bukan bahasa Inggeris | -| 📊**Kemunduran Didorong Penanda Aras** | Kependaman p95 sebenar daripada pemarkahan kombo suapan permintaan langsung — AutoCombo belajar daripada data sebenar | -| 🔁**Minta Deduplikasi** | Tetingkap penyahduaan berasaskan cincang kandungan — selamat berbilang ejen, menghalang caj pendua | -| 🔌**Strategi Penghala Boleh Pasang** | Antara muka `RouterStrategy` yang boleh diperluas — tambah logik penghalaan tersuai sebagai pemalam | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Ciri | Apa yang Dilakukan | -| -------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Taman Permainan Model** | Halaman papan pemuka untuk menguji mana-mana model secara langsung — pemilih pembekal/model/titik akhir, Editor Monaco, penstriman, batalkan, pemasaan | -| 🔏**Padanan Cap Jari CLI** | Penyusunan pengepala/badan setiap penyedia untuk memadankan tandatangan CLI asli — togol setiap pembekal dalam Tetapan > Keselamatan.**IP proksi anda dipelihara** | -| 🤝**Sokongan ACP (Protokol Pelanggan Agen)** | Penemuan ejen CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 lagi), pemijah proses, titik akhir `/api/acp/agent` | -| 🤖**Papan Pemuka Ejen ACP** | Nyahpepijat › Halaman ejen — grid 14 ejen dengan status pemasangan, versi, borang ejen tersuai untuk sebarang alat CLI. Pengguna**OpenCode**mendapat butang "Muat turun opencode.json" yang menjana secara automatik konfigurasi sedia untuk digunakan dengan semua model yang tersedia. | -| 🔧**Penghalaan `apiFormat` Model Tersuai** | Model tersuai dengan `apiFormat: "respons"` kini dihalakan dengan betul ke penterjemah API Respons | -| 🏢**Pengasingan Ruang Kerja Codex** | Berbilang ruang kerja Codex setiap e-mel — OAuth memisahkan sambungan dengan betul mengikut ID ruang kerja | -| 🔄**Kemas Kini Auto Elektron** | Apl desktop menyemak kemas kini + pemasangan automatik semasa dimulakan semula | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Ciri | Apa yang Dilakukan | -| ---------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**Pelayan MCP (25 alatan)** | Alat IDE/agen melalui 3 pengangkutan: stdio, SSE (`/api/mcp/sse`), HTTP Boleh Strim (`/api/mcp/stream`). 18 teras + 3 memori + 4 alatan kemahiran | -| 🤝**Pelayan A2A (JSON-RPC + SSE)** | Pelaksanaan tugas ejen kepada ejen dengan aliran penyegerakan dan penstriman | -| 🧭**Halaman Titik Akhir Disatukan** | Halaman pengurusan tab dengan tab Endpoint Proxy, MCP, A2A dan Titik Akhir API | -| 🎚️**Servis Dayakan/Lumpuhkan Togol** | Suis HIDUP/MATI untuk MCP dan A2A dengan tetapan tetapan (lalai: MATI) | -| 🛰️**Denyutan Jantung Masa Jalanan MCP** | Status proses sebenar (pid, masa beroperasi, umur degupan jantung, pengangkutan, mod skop) | -| 📋**Jejak Audit MCP** | Log audit boleh ditapis dengan kejayaan/kegagalan dan atribusi utama | -| 🔐**Penguatkuasaan Skop MCP** | 10 kebenaran skop berbutir untuk akses alat terkawal | -| 📡**Pengurusan Kitaran Hayat Tugas A2A** | Senaraikan/tapis tugas, periksa acara/artifak, batalkan tugas yang sedang dijalankan | -| 📋**Penemuan Kad Agen** | `/.well-known/agent.json` untuk penemuan automatik pelanggan | -| 🧪**Protokol E2E Test Harness** | Klien MCP SDK + A2A sebenar mengalir dalam `test:protocols:e2e` | -| ⚙️**Kawalan Operasi** | Tukar kombo, gunakan profil daya tahan, tetapkan semula pemutus dari satu permukaan kawalan | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Ciri | Apa yang Dilakukan | -| ----------------------------------- | ------------------------------------------------------------------------------------ | ----------------------- | -| 🎯**Smart 4-Tier Fallback** | Laluan automatik: Langganan → Kunci API → Murah → Percuma | -| 📊**Penjejakan Kuota Masa Nyata** | Kiraan token langsung + tetapan semula kira detik setiap pembekal | -| 🔄**Terjemahan Format** | OpenAI ↔ Claude ↔ Gemini ↔ Respons dengan penukaran selamat skema | -| 👥**Sokongan Berbilang Akaun** | Berbilang akaun bagi setiap pembekal dengan pemilihan pintar | -| 🔄**Muat Semula Token Auto** | Token OAuth dimuat semula secara automatik dengan cuba semula | -| 🎨**Kombo Tersuai** | 9 strategi mengimbangi + kawalan rantai sandaran | -| 🌐**Penghala Wildcard** | `penyedia/*` penghalaan dinamik | -| 🧠**Kawalan Belanjawan Berfikir** | Had penaakulan laluan, auto, tersuai dan adaptif | -| 🔀**Alias ​​Model** | Pengalian model + tersuai terbina dalam dan keselamatan penghijrahan | -| ⚡**Degradasi Latar Belakang** | Halakan tugas latar belakang keutamaan rendah ke model yang lebih murah | -| 🧪**Penghalaan Pintar Sedar Tugas** | Autopilih model mengikut jenis kandungan (pengekodan/penglihatan/analisis/ringkasan) | -| 🔄**Aliran Kerja Ejen A2A** | Pengatur FSM yang menentukan untuk pelaksanaan ejen berbilang langkah berstatus | -| 🔀**Penghalaan Adaptif** | Penggantian strategi dinamik berdasarkan volum token dan kerumitan segera | -| 🎲**Kepelbagaian Pembekal** | Pemarkahan entropi Shannon mengimbangi pengedaran trafik kombo automatik | -| 💬**System Prompt Suntikan** | Kawalan tingkah laku global digunakan secara konsisten | -| 📄**Kesesuaian API Respons** | Sokongan penuh `/v1/respons` untuk Codex dan aliran kerja agen lanjutan | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Ciri | Apa yang Dilakukan | -| ----------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**Penjanaan Imej** | `/v1/images/generations` dengan awan dan hujung belakang tempatan | -| 📐**Pembenaman** | `/v1/embeddings` untuk carian dan saluran paip RAG | -| 🎤**Transkripsi Audio** | `/v1/audio/transkripsi` — 7 pembekal (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), pengesanan autobahasa, sokongan MP4/MP3/WAV | -| 🔊**Teks-ke-Ucapan** | `/v1/audio/speech` — 10 pembekal (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) dengan mesej ralat yang betul | -| 🎬**Penjanaan Video** | `/v1/video/generasi` (aliran kerja ComfyUI + SD WebUI) | -| 🎵**Generasi Muzik** | `/v1/muzik/generasi` (aliran kerja ComfyUI) | -| 🛡️**Kesederhanaan** | `/v1/moderations` semakan keselamatan | -| 🔀**Penyusunan semula** | `/v1/rerank` untuk pemarkahan perkaitan | -| 🔍**Carian Web**🆕 | `/v1/search` — 5 pembekal (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ percuma/bulan, auto-failover, cache | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Ciri | Apa yang Dilakukan | -| ------------------------------------- | ----------------------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**Pemutus Litar** | Perjalanan/pulih setiap model dengan kawalan ambang | -| 🎯**Model Sedar Titik Akhir** | Model tersuai mengisytiharkan titik akhir yang disokong + format API | -| 🛡️**Kawanan Anti Guruh** | Perlindungan Mutex + semaphore pada acara cuba semula/kadar | -| 🧠**Semantik + Cache Tandatangan** | Pengurangan kos/pendaman dengan dua lapisan cache | -| ⚡**Minta Idepotency** | Tetingkap perlindungan pendua | -| 🔒**TLS Fingerprint Spoofing** | Cap jari TLS seperti pelayar —**mengurangkan pengesanan bot dan pembenderaan akaun** | -| 🔏**Padanan Cap Jari CLI** | Padan dengan tandatangan permintaan CLI asli —**mengurangkan risiko larangan sambil mengekalkan IP proksi** | -| 🌐**Penapisan IP** | Kawalan senarai benar/senarai sekat untuk penempatan terdedah | -| 📊**Had Kadar Boleh Diedit** | Had global/peringkat pembekal boleh dikonfigurasikan dengan kegigihan | -| 📉**Degradasi Anggun** | Sandaran keupayaan berbilang lapisan melindungi operasi get laluan teras | -| 📜**Jejak Audit Config** | Penjejakan perubahan berasaskan perbezaan menghalang hanyut operasi dengan pemulangan mudah | -| ⏳**Penyegerakan Kesihatan Pembekal** | Pemantauan tamat tempoh token proaktif mencetuskan makluman sebelum kegagalan kebenaran | -| 🚪**AutoLumpuhkan Akaun Diharamkan** | Pemutus litar operasi mengelak akaun token yang disekat secara kekal secara automatik | -| 🔑**Pengurusan Kunci API + Skop** | Kawalan pengeluaran/putaran kunci dan model/pembekal selamat | -| 👁️**Pendedahan Kunci API Berskop**🆕 | Ikut serta pemulihan kunci API melalui `ALLOW_API_KEY_REVEAL` | -| 🛡️**Dilindungi `/model`** | Gating pengesahan pilihan dan penyembunyian penyedia untuk katalog model | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Ciri | Apa yang Dilakukan | -| ------------------------------------ | ---------------------------------------------------------------- | ---------------------------- | -| 📝**Permintaan + Pembalakan Proksi** | Permintaan/tindak balas penuh dan pengelogan proksi | -| 📉**Log Terperinci Distrim**🆕 | Membina semula aliran muatan SSE dengan bersih ke dalam UI | -| 📋**Papan Pemuka Log Bersatu** | Permintaan, proksi, audit dan paparan konsol dalam satu halaman | -| 🔍**Minta Telemetri** | kependaman p50/p95/p99 dan pengesanan permintaan | -| 🏥**Papan Pemuka Kesihatan** | Masa aktif, keadaan pemutus, sekatan, statistik cache | -| 💰**Penjejakan Kos** | Kawalan belanjawan dan keterlihatan harga setiap model | -| 📈**Penggambaran Analitik** | Cerapan penggunaan model/pembekal dan pandangan arah aliran | -| 🧪**Rangka Kerja Penilaian** | Ujian set emas dengan strategi perlawanan boleh dikonfigurasikan | -| 📡**Diagnostik Langsung**🆕 | Pintasan cache semantik untuk ujian langsung kombo yang tepat | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Ciri | Apa yang Dilakukan | -| ------------------------------- | ------------------------------------------------------------------------------ | --------------------- | -| 🌐**Kerahkan Di Mana-mana** | Localhost, VPS, Docker, persekitaran Cloud | -| 🚇**Terowong Cloudflare**🆕 | Penyepaduan Terowong Pantas satu klik daripada papan pemuka | -| 🔑**Penapisan Model Kunci API** | Respons asli /v1/models ditapis melalui peranan konteks Pembawa yang diberikan | -| ⚡**Pintas Cache Pintar** | Heuristik TTL yang boleh dikonfigurasikan dan kawalan pengambilan semula paksa | -| 🔄**Sandaran/Pulihkan** | Eksport/import dan aliran pemulihan bencana | -| 🧙**Onboarding Wizard** | Persediaan berpandu jalan pertama | -| 🔧**Papan Pemuka Alat CLI** | Persediaan satu klik untuk alat pengekodan popular | -| 🎮**Taman Permainan Model** | Uji mana-mana pembekal/model/titik akhir daripada papan pemuka | -| 🔏**Togol Cap Jari CLI** | Padanan cap jari setiap pembekal dalam Tetapan > Keselamatan | -| 🌐**i18n (30 bahasa)** | Papan pemuka penuh + sokongan bahasa dokumen dengan liputan RTL | -| 🧹**Kosongkan Semua Model** | Pembersihan senarai model satu klik dalam butiran pembekal | -| 👁️**Kawalan Bar Sisi**🆕 | Sembunyikan komponen dan penyepaduan daripada Tetapan Rupa | -| 📋**Templat Isu** | Templat GitHub standard untuk pepijat dan ciri | -| 📂**Direktori Data Tersuai** | `DATA_DIR` menimpa lokasi storan | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1292,103 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -Apabila kuota, kadar atau kesihatan gagal, OmniRoute secara automatik beralih ke calon seterusnya tanpa penukaran manual.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A boleh ditemui dalam UI dan dokumen (tidak tersembunyi) -- API status protokol mendedahkan data operasi langsung (`/api/mcp/*`, `/api/a2a/*`) -- Papan pemuka termasuk tindakan untuk operasi hari ke-2 (togol kombo, penetapan semula pemutus, pembatalan tugas)#### Translator + validation workflow +#### Protocol management that is visible and operable -Kawasan Penterjemah termasuk: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Taman Permainan**: meminta semakan transformasi -**Penguji Sembang**: permintaan penuh/tindak balas pergi balik -**Bangku Ujian**: berbilang kes dalam satu larian -**Pemantau Langsung**: paparan trafik masa nyata +#### Translator + validation workflow -Tambahan pengesahan protokol dengan pelanggan sebenar melalui `npm run test:protocols:e2e`. +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Rujukan alat, konfigurasi IDE dan contoh klien +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A Server README](src/lib/a2a/README.md)**— Kemahiran, kaedah JSON-RPC, penstriman dan kitaran hayat tugas## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute termasuk rangka kerja penilaian terbina dalam untuk menguji kualiti tindak balas LLM terhadap set emas. Aksesnya melalui**Analytics → Evals**dalam papan pemuka.### Built-in Golden Set +## 🧪 Evaluations (Evals) -"Set Emas OmniRoute" yang dipramuat mengandungi kes ujian untuk: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Salam, matematik, geografi, penjanaan kod -- Pematuhan format JSON, terjemahan, penjanaan turun nilai -- Penolakan keselamatan (kandungan berbahaya), pengiraan, logik boolean### Evaluation Strategies +### Built-in Golden Set -| Strategi | Penerangan | Contoh | -| ------------- | --------------------------------------------------------------------- | --------------------------------- | --- | -| `tepat` | Output mesti sepadan dengan tepat | `"4"` | -| `mengandungi` | Output mesti mengandungi subrentetan (tidak peka huruf besar-besaran) | `"Paris"` | -| `regex` | Output mesti sepadan dengan corak regex | `"1.*2.*3"` | -| `adat` | Fungsi JS tersuai mengembalikan benar/salah | `(output) => output.panjang > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) - +
🧩 MCP Setup (Model Context Protocol) -Mulakan pengangkutan MCP dalam mod stdio:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Aliran pengesahan yang disyorkan: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Sambungkan klien MCP anda melalui stdio. -2. Jalankan `omniroute_get_health`. -3. Jalankan `omniroute_list_combos`. -4. Buka `/papan pemuka/mcp` untuk mengesahkan degupan jantung, aktiviti dan audit. +Useful APIs for automation: -API berguna untuk automasi: +- `GET /api/mcp/status` +- `GET /api/mcp/tools` +- `GET /api/mcp/audit` +- `GET /api/mcp/audit/stats` -- `DAPATKAN /api/mcp/status` -- `DAPATKAN /api/mcp/tools` -- `DAPATKAN /api/mcp/audit` -- `DAPATKAN /api/mcp/audit/stats`
+ - -🤝 Persediaan A2A (Agent2Agent) +
+🤝 A2A Setup (Agent2Agent) -Temui ejen:```bash +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Hantar tugasan:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` +Manage lifecycle: -Urus kitaran hayat: - -- `DAPATKAN /api/a2a/status` -- `DAPATKAN /api/a2a/tasks` -- `DAPATKAN /api/a2a/tasks/:id` +- `GET /api/a2a/status` +- `GET /api/a2a/tasks` +- `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -UI Operasi: +Operational UI: -- `/papan pemuka/a2a` untuk pemerhatian tugas/keadaan/strim dan tindakan asap
+- `/dashboard/a2a` for task/state/stream observability and smoke actions - -🧪 Pengesahan protokol hujung ke hujung + -Sahkan kedua-dua protokol dengan pelanggan sebenar:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Ini mengesahkan: +This verifies: -- Sambung/senarai/panggilan klien MCP SDK -- Penemuan A2A/hantar/strim/dapat/batal -- Semak silang data dalam audit MCP dan API pengurusan tugas A2A
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs - -💳 Pembekal Langganan### Claude Code (Pro/Max) + + +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1401,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Petua Pro:**Gunakan Opus untuk tugas yang rumit, Sonnet untuk kelajuan. OmniRoute menjejaki kuota setiap model!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1415,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Setiap akaun Codex kini mempunyai togol dasar dalam `Papan Pemuka -> Pembekal`: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5j` (HIDUP/MATI): menguatkuasakan dasar ambang tetingkap 5 jam. -- `Mingguan` (HIDUP/MATI): menguatkuasakan dasar ambang tetingkap mingguan. -- Tingkah laku ambang: apabila tetingkap yang didayakan mencapai >=90% penggunaan, akaun itu dilangkau. -- Tingkah laku putaran: Laluan OmniRoute ke akaun Codex yang layak seterusnya secara automatik. -- Tetapkan semula tingkah laku: apabila masa pembekal `resetAt` berlalu, akaun menjadi layak semula secara automatik. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Senario: +Scenarios: -- `5j HIDUP` + `Mingguan HIDUP`: akaun dilangkau apabila mana-mana tetingkap mencapai ambang. -- `5j OFF` + `Weekly ON`: hanya penggunaan mingguan boleh menyekat akaun. -- `5j ON` + `Mingguan OFF`: hanya penggunaan 5 jam boleh menyekat akaun. -- `resetAt` lulus: akaun memasuki semula putaran secara automatik (tiada manual didayakan semula).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1440,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Nilai Terbaik:**Peringkat percuma yang besar! Gunakan ini sebelum peringkat berbayar.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1455,71 +1662,91 @@ Models:
- -🔑 API Key Providers### NVIDIA NIM (FREE developer access — 70+ models) +
+🔑 API Key Providers -1. Daftar: [build.nvidia.com](https://build.nvidia.com) -2. Dapatkan kunci API percuma (1000 kredit inferens disertakan) -3. Papan Pemuka → Tambah Pembekal → NVIDIA NIM: - - Kunci API: `nvapi-your-key` +### NVIDIA NIM (FREE developer access — 70+ models) -**Model:**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` dan 50+ lagi +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Petua Pro:**API serasi OpenAI — berfungsi dengan lancar dengan terjemahan format OmniRoute!### DeepSeek +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -1. Daftar: [platform.deepseek.com](https://platform.deepseek.com) -2. Dapatkan kunci API -3. Papan Pemuka → Tambah Pembekal → DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -**Model:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +### DeepSeek -1. Daftar: [console.groq.com](https://console.groq.com) -2. Dapatkan kunci API (termasuk peringkat percuma) -3. Papan Pemuka → Tambah Pembekal → Groq +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -**Model:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Petua Pro:**Inferens sangat pantas — terbaik untuk pengekodan masa nyata!### OpenRouter (100+ Models) +### Groq (Free Tier Available!) -1. Daftar: [openrouter.ai](https://openrouter.ai) -2. Dapatkan kunci API -3. Papan Pemuka → Tambah Pembekal → OpenRouter +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -**Model:**Akses 100+ model daripada semua pembekal utama melalui kunci API tunggal. +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Tingkah laku papan pemuka:**Model OpenRouter diurus daripada**Model Tersedia**. Tambah, import dan autosegerak secara manual semua mengemas kini senarai yang sama.
+**Pro Tip:** Ultra-fast inference — best for real-time coding! - -💰 Penyedia Murah (Sandaran)### GLM-4.7 (Daily reset, $0.6/1M) +### OpenRouter (100+ Models) -1. Daftar: [Zhipu AI](https://open.bigmodel.cn/) -2. Dapatkan kunci API daripada Pelan Pengekodan -3. Papan Pemuka → Tambah Kunci API: - - Pembekal: `glm` - - Kunci API: `kunci-anda` +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -**Kegunaan:**`glm/glm-4.7` +**Models:** Access 100+ models from all major providers through a single API key. -**Petua Pro:**Pelan Pengekodan menawarkan kuota 3× pada kos 1/7! Tetapkan semula setiap hari 10:00 AM.### MiniMax M2.1 (5h reset, $0.20/1M) +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -1. Daftar: [MiniMax](https://www.minimax.io/) -2. Dapatkan kunci API -3. Papan Pemuka → Tambah Kunci API + -**Kegunaan:**`minimax/MiniMax-M2.1` +
+💰 Cheap Providers (Backup) -**Petua Pro:**Pilihan termurah untuk konteks panjang (token 1M)!### Kimi K2 ($9/month flat) +### GLM-4.7 (Daily reset, $0.6/1M) -1. Langgan: [Moonshot AI](https://platform.moonshot.ai/) -2. Dapatkan kunci API -3. Papan Pemuka → Tambah Kunci API +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**Gunakan:**`kimi/kimi-terbaru` +**Use:** `glm/glm-4.7` -**Petua Pro:**Tetap $9/bulan untuk 10J token = $0.90/1J kos efektif!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. - -🆓 Pembekal PERCUMA (Sandaran Kecemasan)### Qoder (5 FREE models via OAuth) +### MiniMax M2.1 (5h reset, $0.20/1M) + +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `minimax/MiniMax-M2.1` + +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1560,8 +1787,10 @@ Models:
- -🎨 Cipta Kombo### Example 1: Maximize Subscription → Cheap Backup +
+🎨 Create Combos + +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1589,8 +1818,10 @@ Cost: $0 forever!
- -🔧 Penyepaduan CLI### Cursor IDE +
+🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1601,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Gunakan halaman**CLI Tools**dalam papan pemuka untuk konfigurasi satu klik atau edit `~/.claude/settings.json` secara manual.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1612,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Pilihan 1 — Papan Pemuka (disyorkan):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Pilihan 2 — Manual:**Edit `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1629,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Nota:**OpenClaw hanya berfungsi dengan OmniRoute tempatan. Gunakan `127.0.0.1` dan bukannya `localhost` untuk mengelakkan isu resolusi IPv6.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1643,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Langkah 1:**Tambahkan OmniRoute sebagai pembekal tersuai:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Langkah 2:**Cipta/edit `opencode.json` dalam akar projek anda:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1669,117 +1909,130 @@ opencode } } } -```` +``` -**Langkah 3:**Pilih model dalam OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Petua:**Tambahkan sebarang model yang tersedia dalam titik akhir OmniRoute `/v1/models` anda ke bahagian `models`. Gunakan format `provider/model-id` daripada papan pemuka OmniRoute anda.
+ --- ## Penyelesaian Masalah - -Klik untuk mengembangkan panduan penyelesaian masalah +
+Click to expand troubleshooting guide -**"Model bahasa tidak memberikan mesej"** +**"Language model did not provide messages"** -- Kuota pembekal habis → Semak penjejak kuota papan pemuka -- Penyelesaian: Gunakan sandaran kombo atau tukar kepada peringkat yang lebih murah +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**Penghadan kadar** +**Rate limiting** -- Kuota langganan habis → Sandar kepada GLM/MiniMax -- Tambah kombo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**Token OAuth tamat tempoh** +**OAuth token expired** -- Dikemas semula secara automatik oleh OmniRoute -- Jika isu berterusan: Papan Pemuka → Pembekal → Sambung semula +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Kos tinggi** +**High costs** -- Semak statistik penggunaan dalam Papan Pemuka → Kos -- Tukar model utama kepada GLM/MiniMax -- Gunakan peringkat percuma (Gemini CLI, Qoder) untuk tugasan yang tidak kritikal +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Port papan pemuka/API salah** +**Dashboard/API ports are wrong** -- `PORT` ialah port asas kanonik (dan port API secara lalai) -- `API_PORT` mengatasi hanya pendengar API yang serasi dengan OpenAI -- `DASHBOARD_PORT` mengatasi hanya papan pemuka/pendengar Next.js -- Tetapkan `NEXT_PUBLIC_BASE_URL` pada papan pemuka/URL awam anda (untuk panggilan balik OAuth) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Ralat penyegerakan awan** +**Cloud sync errors** -- Sahkan mata `BASE_URL` pada contoh berjalan anda -- Sahkan mata `CLOUD_URL` ke titik akhir awan anda yang dijangkakan -- Pastikan nilai `NEXT_PUBLIC_*` sejajar dengan nilai sebelah pelayan +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Log masuk pertama tidak berfungsi** +**First login not working** -- Tandai `INITIAL_PASSWORD` dalam `.env` -- Jika tidak ditetapkan, kata laluan sandaran ialah `123456` +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Tiada log permintaan** +**No request logs** -- Artifak permintaan ditulis ke `DATA_DIR/call_logs/` sebagai satu fail JSON setiap permintaan -- Dayakan tangkapan saluran paip dari Papan Pemuka → Log → Log Permintaan jika anda memerlukan muatan setiap peringkat yang terperinci -- Tetapkan `APP_LOG_TO_FILE=true` jika anda juga mahu log konsol aplikasi dalam `logs/application/app.log` -- Laraskan `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` dan `CALL_LOG_MAX_ENTRIES` mengikut keperluan +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Ujian sambungan menunjukkan "Tidak sah" untuk pembekal yang serasi dengan OpenAI** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Banyak pembekal tidak mendedahkan titik akhir `/models` -- OmniRoute v1.0.6+ termasuk pengesahan sandaran melalui pelengkapan sembang -- Pastikan URL asas mengandungi akhiran `/v1`### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ Penting untuk pengguna yang menjalankan OmniRoute pada VPS, Docker atau mana-mana pelayan jauh**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -Pembekal**Antigravity**dan**Gemini CLI**menggunakan**Google OAuth 2.0**. Google memerlukan `redirect_uri` dalam aliran OAuth supaya betul-betul sepadan dengan salah satu URI pradaftar dalam Google Cloud Console apl. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -Bukti kelayakan OAuth yang digabungkan dalam OmniRoute didaftarkan**untuk `localhost` sahaja**. Apabila anda mengakses OmniRoute pada pelayan jauh (cth. `https://omniroute.myserver.com`), Google menolak pengesahan dengan:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Anda perlu membuat**ID Klien OAuth 2.0**dalam Google Cloud Console dengan URI pelayan anda.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Buka Google Cloud Console** +#### Step-by-step -Pergi ke: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Buat ID Pelanggan OAuth 2.0 baharu** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Klik**"+ Cipta Bukti kelayakan"**→**"ID klien OAuth"** -- Jenis aplikasi:**"Aplikasi web"** -- Nama: apa sahaja yang anda suka (cth. `OmniRoute Remote`) +**2. Create a new OAuth 2.0 Client ID** -**3. Tambah URI Ubah Hala Dibenarkan** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -Dalam medan**"URI halauan dibenarkan"**, tambahkan:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Gantikan `your-server.com` dengan domain atau IP pelayan anda (termasuk port jika perlu, cth. `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. Simpan dan salin bukti kelayakan** +After creating, Google will show the **Client ID** and **Client Secret**. -Selepas membuat, Google akan menunjukkan**ID Pelanggan**dan**Rahsia Pelanggan**. +**5. Set environment variables** -**5. Tetapkan pembolehubah persekitaran** +In your `.env` (or Docker environment variables): -Dalam `.env` anda (atau pembolehubah persekitaran Docker):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1788,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Mulakan semula OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Cuba sambung semula** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Papan pemuka → Pembekal → Antigraviti (atau Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google kini akan mengubah hala dengan betul ke `https://your-server.com/callback`.--- +--- #### Temporary workaround (without custom credentials) -Jika anda tidak mahu menyediakan bukti kelayakan anda sendiri sekarang, anda masih boleh menggunakan**aliran URL manual**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute membuka URL kebenaran Google -2. Selepas memberi kebenaran, Google cuba mengubah hala ke `localhost` (yang gagal pada pelayan jauh) -3.**Salin URL penuh**dari bar alamat penyemak imbas anda (walaupun halaman tidak dimuatkan) -4. Tampalkan URL tersebut ke dalam medan yang ditunjukkan dalam modal sambungan OmniRoute -5. Klik**"Sambung"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Ini berfungsi kerana kod kebenaran dalam URL adalah sah tidak kira sama ada halaman ubah hala dimuatkan.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. - -🇧🇷 Versão em Português#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Ia membuktikan**Antigraviti**dan**Gemini CLI**menggunakan**Google OAuth 2.0**untuk autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja**exatamente**das URIs pre-cadastradas no Google Cloud Console to aplicativo. +
+🇧🇷 Versão em Português -Sebagai kredenciais OAuth embutidas no OmniRoute estão cadastradas**apenas untuk `localhost`**. Anda boleh mengakses OmniRoute em um servidor remoto (cth: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Você precisa criar um**OAuth 2.0 Client ID**no Google Cloud Console com a URI do seu servidor.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Akses ke Konsol Awan Google** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -**2. Crie um novo ID Pelanggan OAuth 2.0** +**2. Crie um novo OAuth 2.0 Client ID** -- Klik em**"+ Cipta Bukti Kelayakan"**→**"ID klien OAuth"** -- Tipo de aplicativo:**"Aplikasi web"** -- Nama: escolha qualquer nome (cth: `OmniRoute Remote`) +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Tambah sebagai URI Ubah Hala Dibenarkan** +**3. Adicione as Authorized Redirect URIs** -Tiada**"URI ubah hala yang dibenarkan"**, tambahan:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Gantikan `seu-servidor.com` pelo domínio ou IP do seu servidor (termasuk porta se necessário, cth: `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. Simpan dan salin sebagai kredensia** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Após criar, o Google mostrará o**ID Pelanggan**e o**Rahsia Pelanggan**. +**5. Configure as variáveis de ambiente** -**5. Konfigurasikan sebagai variáveis de ambiente** +No seu `.env` (ou nas variáveis de ambiente do Docker): -No seu `.env` (ou nas variáveis de ambiente do Docker):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1867,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reinicie o OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute - -```` +``` **7. Tente conectar novamente** -Papan pemuka → Pembekal → Antigraviti (ou Gemini CLI) → OAuth +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Agora o Google redirectionará corretamente para `https://seu-servidor.com/callback` dan autenticação funcionará.--- +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. + +--- #### Workaround temporário (sem configurar credenciais próprias) -Jika anda ingin mendapatkan credenciais próprias agora, ada kemungkinan penggunaan atau fluks**manual de URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. OmniRoute abrirá a URL de authorização do Google -2. Após você authorizar, o Google tentará redirectionar para `localhost` (que falha no servidor remoto) -3.**Salin URL yang lengkap**pada penyemak imbas barra de endeço do seu (mesmo que a página não carregue) +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) 4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute -5. Klik em**"Sambung"** +5. Clique em **"Connect"** -> Penyelesaian ini berfungsi sebagai kodigo de autorização na URL adalah bebas untuk mengubah hala mengikut arahan atau tidak.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1905,64 +2171,73 @@ Jika anda ingin mendapatkan credenciais próprias agora, ada kemungkinan penggun ## 🛠️ Tech Stack - -Klik untuk mengembangkan butiran tindanan teknologi +
+Click to expand tech stack details --**Waktu Jalan**: Node.js 18–22 LTS (⚠️ Node.js 24+**tidak disokong**— perduaan asli `better-sqlite3` tidak serasi) --**Bahasa**: TypeScript 5.9 —**100% TypeScript**merentas `src/` dan `open-sse/` (sifar `mana-mana` dalam modul teras sejak v2.0) --**Kerangka**: Next.js 16 + React 19 + Tailwind CSS 4 --**Pangkalan Data**: LowDB (JSON) + SQLite (keadaan domain + log proksi + audit MCP + keputusan penghalaan) --**Skema**: Zod (pengesahan I/O alat MCP, kontrak API) --**Protokol**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Penstriman**: Acara Dihantar Pelayan (SSE) --**Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization --**Ujian**: Pelari ujian Node.js + Vitest (900+ ujian termasuk unit, penyepaduan, E2E) --**CI/CD**: GitHub Actions (auto npm publish + Docker Hub pada keluaran) --**Tapak web**: [omniroute.online](https://omniroute.online) --**Pakej**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Ketahanan**: Pemutus litar, pengunduran eksponen, kumpulan anti-gemuruh, penipuan TLS, penyembuhan diri kombo automatik
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Dokumentasi -| Dokumen | Penerangan | +| Document | Description | | ---------------------------------------------- | --------------------------------------------------- | -| [Panduan Pengguna](docs/USER_GUIDE.md) | Pembekal, kombo, penyepaduan CLI, penggunaan | -| [Rujukan API](docs/API_REFERENCE.md) | Semua titik akhir dengan contoh | -| [Pelayan MCP](open-sse/mcp-server/README.md) | 16 alatan MCP, konfigurasi IDE, pelanggan Python/TS/Go | -| [Pelayan A2A](src/lib/a2a/README.md) | JSON-RPC 2.0 protokol, kemahiran, penstriman, mgmt tugas | -| [Enjin Auto-Kombo](docs/auto-combo.md) | Pemarkahan 6 faktor, pek mod, penyembuhan diri | -| [Penyelesaian masalah](docs/PENYELESAIAN MASALAH.md) | Masalah dan penyelesaian biasa | -| [Seni Bina](docs/ARCHITECTURE.md) | Seni bina sistem dan dalaman | -| [Menyumbang](MENYUMBANG.md) | Persediaan pembangunan dan garis panduan | -| [Spesifikasi OpenAPI](docs/openapi.yaml) | Spesifikasi OpenAPI 3.0 | -| [Dasar Keselamatan](SECURITY.md) | Pelaporan kerentanan dan amalan keselamatan | -| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Panduan lengkap: Persediaan VM + nginx + Cloudflare | -| [Galeri Ciri](docs/FEATURES.md) | Lawatan papan pemuka visual dengan tangkapan skrin | -| [Senarai Semak Keluaran](docs/RELEASE_CHECKLIST.md) | Langkah pengesahan prakeluaran |--- +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute mempunyai**210+ ciri yang dirancang**merentas berbilang fasa pembangunan. Berikut adalah bidang utama: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Kategori | Planned Features | Sorotan | -| ---------------------------- | ---------------- | ------------------------------------------------------------------------------------ | -| 🧠**Penghalaan & Perisikan**| 25+ | Penghalaan kependaman terendah, penghalaan berasaskan teg, kuota prapenerbangan, pemilihan akaun P2C | -| 🔒**Keselamatan & Pematuhan**| 20+ | Pengerasan SSRF, penyelubungan kelayakan, had kadar setiap titik akhir, skop kunci pengurusan | -| 📊**Kebolehlihatan**| 15+ | Penyepaduan OpenTelemetry, pemantauan kuota masa nyata, penjejakan kos setiap model | -| 🔄**Integrasi Pembekal**| 20+ | Pendaftaran model dinamik, penyejukan pembekal, Codex berbilang akaun, penghuraian kuota Copilot | -| ⚡**Prestasi**| 15+ | Lapisan cache dwi, ​​cache gesaan, cache respons, penstriman keepalive, API kelompok | -| 🌐**Ekosistem**| 10+ | API WebSocket, konfigurasi hot-reload, kedai konfigurasi teragih, mod komersial |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**OpenCode Integration**— Sokongan pembekal asli untuk IDE pengekodan AI OpenCode -- 🔗**Pengintegrasian TRAE**— Sokongan penuh untuk rangka kerja pembangunan TRAE AI -- 📦**API Kelompok**— Pemprosesan kelompok tak segerak untuk permintaan pukal -- 🎯**Penghalaan Berasaskan Teg**— Permintaan laluan berdasarkan teg tersuai dan metadata -- 💰**Strategi Kos Terendah**— Pilih pembekal yang tersedia paling murah secara automatik +### 🔜 Coming Soon -> 📝 Spesifikasi ciri penuh tersedia dalam [`docs/new-features/`](docs/new-features/) (217 spesifikasi terperinci)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1970,18 +2245,20 @@ OmniRoute mempunyai**210+ ciri yang dirancang**merentas berbilang fasa pembangun ### How to Contribute -1. Garpu repositori -2. Cipta cawangan ciri anda (`git checkout -b feature/amazing-feature`) -3. Komit perubahan anda (`git commit -m 'Tambah ciri menakjubkan'`) -4. Tekan ke cawangan (`git push origin feature/amazing-feature`) -5. Buka Permintaan Tarik +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Lihat [CONTRIBUTING.md](CONTRIBUTING.md) untuk mendapatkan garis panduan terperinci.### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1993,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Terima kasih khas kepada**[9router](https://github.com/decolua/9router)**oleh**[decolua](https://github.com/decolua)**— projek asal yang mengilhamkan fork ini. OmniRoute membina asas yang luar biasa itu dengan ciri tambahan, API berbilang modal dan penulisan semula TypeScript penuh. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Terima kasih khas kepada**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— pelaksanaan Go asal yang mengilhamkan port JavaScript ini.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Lesen -Lesen MIT - lihat [LESEN](LESEN) untuk mendapatkan butiran.--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/ms/docs/ARCHITECTURE.md b/docs/i18n/ms/docs/ARCHITECTURE.md index 32794fb541..53d2247396 100644 --- a/docs/i18n/ms/docs/ARCHITECTURE.md +++ b/docs/i18n/ms/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Terakhir dikemas kini: 2026-03-28_## Executive Summary -OmniRoute ialah gerbang laluan dan papan pemuka penghalaan AI tempatan yang dibina pada Next.js. -Ia menyediakan satu titik akhir serasi OpenAI (`/v1/*`) dan mengarahkan trafik merentasi berbilang penyedia huluan dengan terjemahan, sandaran, muat semula token dan penjejakan penggunaan. -Keupayaan teras: +_Last updated: 2026-03-28_ -- Permukaan API serasi OpenAI untuk CLI/alat (28 pembekal) -- Permintaan/tindak balas terjemahan merentas format pembekal -- Model kombo mundur (jujukan berbilang model) -- Saling balik peringkat akaun (berbilang akaun setiap pembekal) -- Pengurusan sambungan pembekal kunci OAuth + API -- Penjanaan benam melalui `/v1/embeddings` (6 pembekal, 9 model) -- Penjanaan imej melalui `/v1/images/generations` (4 pembekal, 9 model) -- Penghuraian teg Fikir (`...`) untuk model penaakulan -- Pembersihan tindak balas untuk keserasian OpenAI SDK yang ketat -- Normalisasi peranan (pembangun→sistem, sistem→pengguna) untuk keserasian silang penyedia -- Penukaran output berstruktur (json_schema → Gemini responseSchema) -- Kegigihan setempat untuk pembekal, kunci, alias, kombo, tetapan, harga -- Penjejakan penggunaan/kos dan pengelogan permintaan -- Penyegerakan awan pilihan untuk penyegerakan berbilang peranti/keadaan -- Senarai dibenarkan/senarai sekatan IP untuk kawalan akses API -- Pengurusan belanjawan berfikir (laluan/auto/tersuai/adaptif) -- Suntikan segera sistem global -- Penjejakan sesi dan cap jari -- Pengehadan kadar dipertingkatkan setiap akaun dengan profil khusus pembekal -- Corak pemutus litar untuk daya tahan pembekal -- Perlindungan kumpulan anti-gemuruh dengan penguncian mutex -- Cache penyahduplikasi permintaan berasaskan tandatangan -- Lapisan domain: ketersediaan model, peraturan kos, dasar sandaran, dasar sekat keluar -- Kegigihan keadaan domain (cache tulis-melalui SQLite untuk sandaran, belanjawan, sekatan, pemutus litar) -- Enjin dasar untuk penilaian permintaan terpusat (kunci → belanjawan → sandaran) -- Minta telemetri dengan pengagregatan kependaman p50/p95/p99 -- ID Korelasi (X-Request-Id) untuk pengesanan hujung ke hujung -- Pengelogan audit pematuhan dengan memilih keluar setiap kunci API -- Rangka kerja Eval untuk jaminan kualiti LLM -- Papan pemuka UI Ketahanan dengan status pemutus litar masa nyata -- Pembekal OAuth modular (12 modul individu di bawah `src/lib/oauth/providers/`) +## Executive Summary -Model masa jalan utama: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- Laluan apl Next.js di bawah `src/app/api/*` laksanakan kedua-dua API papan pemuka dan API keserasian -- SSE/tera laluan yang dikongsi dalam `src/sse/*` + `open-sse/*` mengendalikan pelaksanaan pembekal, terjemahan, penstriman, sandaran dan penggunaan## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Masa jalan gerbang tempatan -- API pengurusan papan pemuka -- Pengesahan pembekal dan penyegaran token -- Minta terjemahan dan penstriman SSE -- Keadaan setempat + kegigihan penggunaan -- Orkestrasi penyegerakan awan pilihan### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Pelaksanaan perkhidmatan awan di belakang `NEXT_PUBLIC_CLOUD_URL` -- Pembekal SLA/pesawat kawalan di luar proses tempatan -- Perduaan CLI luaran sendiri (Claude CLI, Codex CLI, dll.)## Dashboard Surface (Current) +### Out of Scope -Halaman utama di bawah `src/app/(dashboard)/dashboard/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/papan pemuka` — permulaan pantas + gambaran keseluruhan pembekal -- `/papan pemuka/titik akhir` — proksi titik akhir + tab titik akhir MCP + A2A + API -- `/papan pemuka/penyedia` — sambungan pembekal dan bukti kelayakan -- `/papan pemuka/kombo` — strategi kombo, templat, peraturan penghalaan model -- `/papan pemuka/kos` — pengagregatan kos dan keterlihatan harga -- `/papan pemuka/analisis` — analitik dan penilaian penggunaan -- `/dashboard/limits` — kawalan kuota/kadar -- `/dashboard/cli-tools` — CLI onboarding, pengesanan masa jalan, penjanaan konfigurasi -- `/papan pemuka/ejen` — ejen ACP dikesan + pendaftaran ejen tersuai -- `/papan pemuka/media` — imej/video/taman permainan muzik -- `/dashboard/search-tools` — ujian dan sejarah pembekal carian -- `/papan pemuka/kesihatan` — masa hidup, pemutus litar, had kadar -- `/papan pemuka/log` — log permintaan/proksi/audit/konsol -- `/papan pemuka/tetapan` — tab tetapan sistem (umum, penghalaan, lalai kombo, dsb.) -- `/dashboard/api-manager` — Kitaran hayat kunci API dan kebenaran model## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Direktori utama: +Main directories: -- `src/app/api/v1/*` dan `src/app/api/v1beta/*` untuk API keserasian -- `src/app/api/*` untuk API pengurusan/konfigurasi -- Seterusnya menulis semula dalam peta `next.config.mjs` `/v1/*` kepada `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Laluan keserasian penting: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` — termasuk model tersuai dengan `custom: true` -- `src/app/api/v1/embeddings/route.ts` — penjanaan benam (6 pembekal) -- `src/app/api/v1/images/generations/route.ts` — penjanaan imej (4+ penyedia termasuk Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — sembang per-pembekal khusus -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — pembenaman per-pembekal khusus -- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — imej setiap pembekal khusus +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` - `src/app/api/v1beta/models/[...path]/route.ts` -Domain pengurusan: +Management domains: -- Pengesahan/tetapan: `src/app/api/auth/*`, `src/app/api/settings/*` -- Pembekal/sambungan: `src/app/api/providers*` -- Nod pembekal: `src/app/api/provider-nodes*` -- Model tersuai: `src/app/api/provider-models` (GET/POST/DELETE) -- Katalog model: `src/app/api/models/route.ts` (GET) -- Konfigurasi proksi: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- Kekunci/alias/kombo/harga: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- Penggunaan: `src/app/api/usage/*` -- Segerak/awan: `src/app/api/sync/*`, `src/app/api/cloud/*` -- Pembantu perkakas CLI: `src/app/api/cli-tools/*` -- Penapis IP: `src/app/api/settings/ip-filter` (GET/PUT) -- Belanjawan berfikir: `src/app/api/settings/thinking-budget` (GET/PUT) -- Gesaan sistem: `src/app/api/settings/system-prompt` (GET/PUT) -- Sesi: `src/app/api/sessions` (GET) -- Had kadar: `src/app/api/rate-limits` (GET) -- Ketahanan: `src/app/api/resilience` (GET/PATCH) — profil pembekal, pemutus litar, keadaan had kadar -- Tetapan semula daya tahan: `src/app/api/resilience/set semula` (POST) — set semula pemutus + cooldown -- Statistik cache: `src/app/api/cache/stats` (DAPAT/DELETE) -- Ketersediaan model: `src/app/api/models/availability` (GET/POST) -- Telemetri: `src/app/api/telemetri/ringkasan` (GET) -- Belanjawan: `src/app/api/usage/budget` (GET/POST) -- Rantaian mundur: `src/app/api/fallback/chains` (DAPATKAN/POST/DELETE) -- Audit pematuhan: `src/app/api/compliance/audit-log` (GET) +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) - Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- Dasar: `src/app/api/policies` (DAPAT/POST)## 2) SSE + Translation Core +- Policies: `src/app/api/policies` (GET/POST) -Modul aliran utama: +## 2) SSE + Translation Core -- Kemasukan: `src/sse/handlers/chat.ts` -- Orkestrasi teras: `open-sse/handlers/chatCore.ts` -- Penyesuai pelaksanaan pembekal: `open-sse/executors/*` -- Konfigurasi pengesanan format/pembekal: `open-sse/services/provider.ts` -- Parse/resolve model: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Logik sandaran akaun: `open-sse/services/accountFallback.ts` -- Pendaftaran terjemahan: `open-sse/translator/index.ts` -- Transformasi strim: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- Pengekstrakan/penormalan penggunaan: `open-sse/utils/usageTracking.ts` -- Penghurai teg Fikir: `open-sse/utils/thinkTagParser.ts` -- Pengendali benam: `open-sse/handlers/embeddings.ts` -- Membenamkan pendaftaran pembekal: `open-sse/config/embeddingRegistry.ts` -- Pengendali penjanaan imej: `open-sse/handlers/imageGeneration.ts` -- Pendaftaran pembekal imej: `open-sse/config/imageRegistry.ts` -- Pembersihan tindak balas: `open-sse/handlers/responseSanitizer.ts` -- Normalisasi peranan: `open-sse/services/roleNormalizer.ts` +Main flow modules: -Perkhidmatan (logik perniagaan): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- Pemilihan/pemarkahan akaun: `open-sse/services/accountSelector.ts` -- Pengurusan kitaran hayat konteks: `open-sse/services/contextManager.ts` -- Penguatkuasaan penapis IP: `open-sse/services/ipFilter.ts` -- Penjejakan sesi: `open-sse/services/sessionManager.ts` -- Minta penyahduaan: `open-sse/services/signatureCache.ts` -- Suntikan segera sistem: `open-sse/services/systemPrompt.ts` -- Pemikiran pengurusan belanjawan: `open-sse/services/thinkingBudget.ts` -- Penghalaan model kad liar: `open-sse/services/wildcardRouter.ts` -- Pengurusan had kadar: `open-sse/services/rateLimitManager.ts` -- Pemutus litar: `open-sse/services/circuitBreaker.ts` +Services (business logic): -Modul lapisan domain: +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- Ketersediaan model: `src/lib/domain/modelAvailability.ts` -- Peraturan/belanjawan kos: `src/lib/domain/costRules.ts` -- Dasar mundur: `src/lib/domain/fallbackPolicy.ts` -- Penyelesai kombo: `src/lib/domain/comboResolver.ts` -- Dasar penguncian: `src/lib/domain/lockoutPolicy.ts` -- Enjin dasar: `src/domain/policyEngine.ts` — kunci keluar berpusat → belanjawan → penilaian mundur -- Katalog kod ralat: `src/lib/domain/errorCodes.ts` -- ID Permintaan: `src/lib/domain/requestId.ts` -- Ambil tamat masa: `src/lib/domain/fetchTimeout.ts` -- Minta telemetri: `src/lib/domain/requestTelemetry.ts` -- Pematuhan/audit: `src/lib/domain/compliance/index.ts` -- Pelari Eval: `src/lib/domain/evalRunner.ts` -- Kegigihan keadaan domain: `src/lib/db/domainState.ts` — SQLite CRUD untuk rantaian sandaran, belanjawan, sejarah kos, keadaan sekat keluar, pemutus litar +Domain layer modules: -Modul pembekal OAuth (12 fail individu di bawah `src/lib/oauth/providers/`): +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -- Indeks pendaftaran: `src/lib/oauth/providers/index.ts` -- Pembekal individu: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts.ts`,`kilocode.ts.`,`kilocode.ts. -- Pembalut nipis: `src/lib/oauth/providers.ts` — eksport semula daripada modul individu## 3) Persistence Layer +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -DB keadaan utama (SQLite): +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -- Infra teras: `src/lib/db/core.ts` (better-sqlite3, migrasi, WAL) -- Eksport semula fasad: `src/lib/localDb.ts` (lapisan keserasian nipis untuk pemanggil) -- fail: `${DATA_DIR}/storage.sqlite` (atau `$XDG_CONFIG_HOME/omniroute/storage.sqlite` apabila ditetapkan, jika tidak `~/.omniroute/storage.sqlite`) -- entiti (jadual + ruang nama KV): providerConnections, providerNodes, modelAliases, combo, apiKeys, tetapan, harga,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +## 3) Persistence Layer -Kegigihan penggunaan: +Primary state DB (SQLite): -- fasad: `src/lib/usageDb.ts` (modul terurai dalam `src/lib/usage/*`) -- Jadual SQLite dalam `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` -- artifak fail pilihan kekal untuk keserasian/nyahpepijat (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- fail JSON lama dipindahkan ke SQLite melalui migrasi permulaan apabila ada +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -DB Keadaan Domain (SQLite): +Usage persistence: -- `src/lib/db/domainState.ts` — Operasi CRUD untuk keadaan domain -- Jadual (dicipta dalam `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- Corak cache tulis-lalu: Peta dalam ingatan adalah berwibawa pada masa jalan; mutasi ditulis serentak kepada SQLite; keadaan dipulihkan daripada DB pada permulaan sejuk## 4) Auth + Security Surfaces +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- Pengesahan kuki papan pemuka: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- Penjanaan/pengesahan kunci API: `src/shared/utils/apiKey.ts` -- Rahsia pembekal kekal dalam entri `providerConnections` -- Sokongan proksi keluar melalui `open-sse/utils/proxyFetch.ts` (env vars) dan `open-sse/utils/networkProxy.ts` (boleh dikonfigurasikan setiap pembekal atau global)## 5) Cloud Sync +Domain State DB (SQLite): -- Penjadual init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Tugas berkala: `src/shared/services/cloudSyncScheduler.ts` -- Tugas berkala: `src/shared/services/modelSyncScheduler.ts` -- Laluan kawalan: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Keputusan mundur didorong oleh `open-sse/services/accountFallback.ts` menggunakan kod status dan heuristik mesej ralat. Penghalaan kombo menambah satu pengawal tambahan: 400s berskop penyedia seperti kegagalan sekatan kandungan huluan dan pengesahan peranan dianggap sebagai kegagalan model tempatan supaya sasaran kombo kemudiannya masih boleh dijalankan.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -Muat semula semasa trafik langsung dilaksanakan di dalam `open-sse/handlers/chatCore.ts` melalui pelaksana `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -Penyegerakan berkala dicetuskan oleh `CloudSyncScheduler` apabila awan didayakan.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -Fail storan fizikal: +Physical storage files: -- DB masa jalan utama: `${DATA_DIR}/storage.sqlite` -- baris log permintaan: `${DATA_DIR}/log.txt` (artifak compat/debug) -- arkib muatan panggilan berstruktur: `${DATA_DIR}/log_panggilan/` -- penterjemah pilihan/permintaan sesi nyahpepijat: `/log/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: API keserasian -- `src/app/api/v1/providers/[provider]/*`: laluan khusus bagi setiap pembekal (sembang, benam, imej) -- `src/app/api/providers*`: penyedia CRUD, pengesahan, ujian -- `src/app/api/provider-nodes*`: pengurusan nod serasi tersuai -- `src/app/api/provider-models`: pengurusan model tersuai (CRUD) -- `src/app/api/models/route.ts`: API katalog model (alias + model tersuai) -- `src/app/api/oauth/*`: OAuth/kod peranti mengalir -- `src/app/api/keys*`: kitaran hayat kunci API tempatan -- `src/app/api/models/alias`: pengurusan alias -- `src/app/api/combos*`: pengurusan kombo sandaran -- `src/app/api/pricing`: penetapan harga menimpa untuk pengiraan kos -- `src/app/api/settings/proxy`: konfigurasi proksi (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: ujian sambungan proksi keluar (POST) -- `src/app/api/usage/*`: penggunaan dan log API -- `src/app/api/sync/*` + `src/app/api/cloud/*`: penyegerakan awan dan pembantu yang menghadap awan -- `src/app/api/cli-tools/*`: penulis/pemeriksa konfigurasi CLI setempat -- `src/app/api/settings/ip-filter`: Senarai dibenarkan/senarai sekat IP (GET/PUT) -- `src/app/api/settings/thinking-budget`: konfigurasi belanjawan token pemikiran (GET/PUT) -- `src/app/api/settings/system-prompt`: gesaan sistem global (GET/PUT) -- `src/app/api/sessions`: penyenaraian sesi aktif (GET) -- `src/app/api/rate-limits`: status had kadar setiap akaun (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: parse permintaan, pengendalian kombo, gelung pemilihan akaun -- `open-sse/handlers/chatCore.ts`: terjemahan, penghantaran pelaksana, pengendalian semula/refresh, persediaan strim -- `open-sse/executors/*`: rangkaian khusus pembekal dan tingkah laku format### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: daftar penterjemah dan orkestrasi -- Minta penterjemah: `open-sse/translator/request/*` -- Penterjemah respons: `open-sse/translator/response/*` -- Pemalar format: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: konfigurasi/keadaan berterusan dan kegigihan domain pada SQLite -- `src/lib/localDb.ts`: eksport semula keserasian untuk modul DB -- `src/lib/usageDb.ts`: sejarah penggunaan/log panggilan fasad di atas jadual SQLite## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Setiap pembekal mempunyai pelaksana khusus yang memanjangkan `BaseExecutor` (dalam `open-sse/executors/base.ts`), yang menyediakan pembinaan URL, pembinaan pengepala, cuba semula dengan backoff eksponen, cangkuk penyegaran semula kelayakan dan kaedah orkestrasi `execute()`. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Pelaksana | Pembekal | Pengendalian Khas | -| ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | --------------------------------------------------------------------------------- | -| `Pelaksana Lalai` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | URL dinamik/konfigurasi pengepala bagi setiap pembekal | -| `Pelaksana Antigraviti` | Antigraviti Google | ID projek/sesi tersuai, Cuba Semula-Selepas menghuraikan | -| `CodexExecutor` | OpenAI Codex | Menyuntik arahan sistem, memaksa usaha penaakulan | -| `Pelaksana Kursor` | IDE kursor | Protokol ConnectRPC, pengekodan Protobuf, tandatangan permintaan melalui checksum | -| `GithubExecutor` | GitHub Copilot | Penyegaran token salinan, pengepala meniru VSCode | -| `KiroExecutor` | AWS CodeWhisperer/Kiro | Format binari AWS EventStream → penukaran SSE | -| `GeminiCLIEexecutor` | Gemini CLI | Kitaran muat semula token Google OAuth | +### Persistence -Semua pembekal lain (termasuk nod serasi tersuai) menggunakan `DefaultExecutor`.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Pembekal | Format | Pengesahan | Strim | Bukan Strim | Token Refresh | API Penggunaan | -| ---------------- | -------------- | --------------------- | ---------------- | ----------- | ------------- | -------------------- | ------------------------------ | -| Claude | claude | Kunci API / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin sahaja | -| Gemini | gemini | Kunci API / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | -| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | -| Antigraviti | antigraviti | OAuth | ✅ | ✅ | ✅ | ✅ API kuota penuh | -| OpenAI | openai | Kunci API | ✅ | ✅ | ❌ | ❌ | -| Codex | openai-respons | OAuth | ✅ terpaksa | ❌ | ✅ | ✅ Had kadar | -| GitHub Copilot | openai | OAuth + Token Copilot | ✅ | ✅ | ✅ | ✅ Gambar kuota | -| Kursor | kursor | Jumlah semak tersuai | ✅ | ✅ | ❌ | ❌ | -| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Had penggunaan | -| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Setiap permintaan | -| Qoder | openai | OAuth (Asas) | ✅ | ✅ | ✅ | ⚠️ Setiap permintaan | -| OpenRouter | openai | Kunci API | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | claude | Kunci API | ✅ | ✅ | ❌ | ❌ | -| DeepSeek | openai | Kunci API | ✅ | ✅ | ❌ | ❌ | -| Groq | openai | Kunci API | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | openai | Kunci API | ✅ | ✅ | ❌ | ❌ | -| Mistral | openai | Kunci API | ✅ | ✅ | ❌ | ❌ | -| Kebingungan | openai | Kunci API | ✅ | ✅ | ❌ | ❌ | -| Bersama AI | openai | Kunci API | ✅ | ✅ | ❌ | ❌ | -| Bunga Api AI | openai | Kunci API | ✅ | ✅ | ❌ | ❌ | -| Serebral | openai | Kunci API | ✅ | ✅ | ❌ | ❌ | -| Cohere | openai | Kunci API | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | openai | Kunci API | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Format sumber yang dikesan termasuk: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. + +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | + +All other providers (including custom compatible nodes) use the `DefaultExecutor`. + +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: - `openai` -- `openai-respons` +- `openai-responses` - `claude` - `gemini` -Format sasaran termasuk: +Target formats include: -- Sembang/Respons OpenAI +- OpenAI chat/Responses - Claude -- Sampul surat Gemini/Gemini-CLI/Antigraviti +- Gemini/Gemini-CLI/Antigravity envelope - Kiro -- Kursor +- Cursor -Terjemahan menggunakan**OpenAI sebagai format hab**— semua penukaran melalui OpenAI sebagai perantaraan:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Terjemahan dipilih secara dinamik berdasarkan bentuk muatan sumber dan format sasaran pembekal. +Additional processing layers in the translation pipeline: -Lapisan pemprosesan tambahan dalam saluran paip terjemahan: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Pembersihan respons**— Menghapuskan medan bukan standard daripada respons format OpenAI (kedua-dua penstriman dan bukan penstriman) untuk memastikan pematuhan SDK yang ketat --**Penormalan peranan**— Menukar `pembangun` → `sistem` untuk sasaran bukan OpenAI; menggabungkan `sistem` → `pengguna` untuk model yang menolak peranan sistem (GLM, ERNIE) --**Think tag extraction**— Menghuraikan `...` blok daripada kandungan ke dalam medan `reasoning_content` --**Output berstruktur**— Menukar OpenAI `response_format.json_schema` kepada `responseMimeType` + `responseSchema` Gemini## Supported API Endpoints +## Supported API Endpoints -| Titik akhir | Format | Pengendali | -| -------------------------------------------------- | ------------------- | ------------------------------------------------------------------- | -| `POST /v1/chat/completions` | Sembang OpenAI | `src/sse/handlers/chat.ts` | -| `POST /v1/message` | Mesej Claude | Pengendali yang sama (dikesan secara automatik) | -| `POST /v1/respons` | Respons OpenAI | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/embeddings` | Pembenaman OpenAI | `open-sse/handlers/embeddings.ts` | -| `DAPATKAN /v1/embeddings` | Penyenaraian model | Laluan API | -| `POST /v1/images/generations` | Imej OpenAI | `open-sse/handlers/imageGeneration.ts` | -| `DAPATKAN /v1/imej/generasi` | Penyenaraian model | Laluan API | -| `POST /v1/providers/{provider}/chat/completions` | Sembang OpenAI | Khusus bagi setiap pembekal dengan pengesahan model | -| `POST /v1/providers/{provider}/embeddings` | Pembenaman OpenAI | Khusus bagi setiap pembekal dengan pengesahan model | -| `POST /v1/providers/{provider}/images/generations` | Imej OpenAI | Khusus bagi setiap pembekal dengan pengesahan model | -| `POST /v1/messages/count_tokens` | Kiraan Token Claude | Laluan API | -| `DAPATKAN /v1/model` | Senarai Model OpenAI | Laluan API (sembang + benam + imej + model tersuai) | -| `DAPATKAN /api/models/catalog` | Katalog | Semua model dikumpulkan mengikut pembekal + jenis | -| `POST /v1beta/models/*:streamGenerateContent` | Gemini asli | Laluan API | -| `DAPATKAN/LETAK/PADAM /api/tetapan/proksi` | Konfigurasi Proksi | Konfigurasi proksi rangkaian | -| `POST /api/settings/proxy/test` | Kesambungan Proksi | Titik akhir ujian kesihatan/ketersambungan proksi | -| `DAPATKAN/POST/DELETE /api/provider-models` | Model Pembekal | Metadata model pembekal menyokong model tersedia tersuai dan terurus |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -Pengendali pintasan (`open-sse/utils/bypassHandler.ts`) memintas permintaan "buang" yang diketahui daripada Claude CLI — ping pemanasan, pengekstrakan tajuk dan kiraan token — dan mengembalikan**tindak balas palsu**tanpa menggunakan token penyedia huluan. Ini dicetuskan hanya apabila `User-Agent` mengandungi `claude-cli`.## Request Logger Pipeline +## Bypass Handler -Logger permintaan (`open-sse/utils/requestLogger.ts`) menyediakan saluran paip pengelogan nyahpepijat 7 peringkat, dilumpuhkan secara lalai, didayakan melalui `ENABLE_REQUEST_LOGS=true`:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Fail ditulis kepada `/logs//` untuk setiap sesi permintaan.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- cooldown akaun pembekal pada ralat sementara/kadar/auth -- sandaran akaun sebelum permintaan gagal -- sandaran model kombo apabila model semasa/laluan pembekal telah habis## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- prasemak dan muat semula dengan mencuba semula untuk pembekal yang boleh dimuat semula -- 401/403 cuba semula selepas percubaan muat semula dalam laluan teras## 3) Stream Safety +## 2) Token Expiry -- pengawal strim sedar putus sambungan -- strim terjemahan dengan siram akhir strim dan pengendalian `[DONE]` -- sandaran anggaran penggunaan apabila metadata penggunaan pembekal tiada## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- ralat penyegerakan muncul tetapi masa jalan tempatan diteruskan -- penjadual mempunyai logik yang mampu mencuba semula, tetapi pelaksanaan berkala pada masa ini memanggil penyegerakan percubaan tunggal secara lalai## 5) Data Integrity +## 3) Stream Safety -- Penghijrahan skema SQLite dan cangkuk naik taraf automatik pada permulaan -- JSON warisan → laluan keserasian penghijrahan SQLite## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Sumber keterlihatan masa jalan: +## 4) Cloud Sync Degradation -- log konsol daripada `src/sse/utils/logger.ts` -- agregat penggunaan setiap permintaan dalam SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- tangkapan muatan terperinci empat peringkat dalam SQLite (`request_detail_logs`) apabila `settings.detailed_logs_enabled=true` -- log masuk status permintaan teks `log.txt` (pilihan/compat) -- log permintaan/terjemahan dalam pilihan di bawah `log/` apabila `ENABLE_REQUEST_LOGS=true` -- titik akhir penggunaan papan pemuka (`/api/usage/*`) untuk penggunaan UI +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -Tangkapan muatan permintaan terperinci menyimpan sehingga empat peringkat muatan JSON bagi setiap panggilan dihalakan: +## 5) Data Integrity -- permintaan mentah diterima daripada pelanggan -- permintaan diterjemahkan sebenarnya dihantar ke hulu -- respons pembekal dibina semula sebagai JSON; respons yang distrim dipadatkan kepada ringkasan akhir ditambah metadata strim -- respons pelanggan akhir dikembalikan oleh OmniRoute; respons yang distrim disimpan dalam bentuk ringkasan padat yang sama## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- Rahsia JWT (`JWT_SECRET`) menjamin pengesahan/penandatanganan kuki sesi papan pemuka -- Tali but kata laluan awal (`INITIAL_PASSWORD`) harus dikonfigurasikan secara eksplisit untuk peruntukan jalan pertama -- Rahsia HMAC kunci API (`API_KEY_SECRET`) menjamin format kunci API tempatan yang dijana -- Rahsia pembekal (kunci/token API) dikekalkan dalam DB tempatan dan harus dilindungi pada peringkat sistem fail -- Titik akhir penyegerakan awan bergantung pada pengesahan kunci API + semantik id mesin## Environment and Runtime Matrix +## Observability and Operational Signals -Pembolehubah persekitaran digunakan secara aktif oleh kod: +Runtime visibility sources: -- Apl/auth: `JWT_SECRET`, `INITIAL_PASSWORD` -- Storan: `DATA_DIR` -- Tingkah laku nod yang serasi: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Penggantian asas storan pilihan (Linux/macOS apabila `DATA_DIR` dinyahset): `XDG_CONFIG_HOME` -- Pencincangan keselamatan: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Pengelogan: `ENABLE_REQUEST_LOGS` -- Penyegerakan/URL awan: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- Proksi keluar: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` dan varian huruf kecil -- Bendera ciri SOCKS5: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- Pembantu platform/masa jalan (bukan konfigurasi khusus apl): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` dan `localDb` berkongsi dasar direktori asas yang sama (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) dengan pemindahan fail lama. -2. `/api/v1/route.ts` mewakilkan kepada pembina katalog bersatu yang sama yang digunakan oleh `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) untuk mengelakkan drift semantik. -3. Permintaan logger menulis tajuk/badan penuh apabila didayakan; anggap direktori log sebagai sensitif. -4. Gelagat awan bergantung pada `NEXT_PUBLIC_BASE_URL` dan kebolehcapaian titik akhir awan yang betul. -5. Direktori `open-sse/` diterbitkan sebagai `@omniroute/open-sse`**pakej ruang kerja npm**. Kod sumber mengimportnya melalui `@omniroute/open-sse/...` (diselesaikan oleh Next.js `transpilePackages`). Laluan fail dalam dokumen ini masih menggunakan nama direktori `open-sse/` untuk konsistensi. -6. Carta dalam papan pemuka menggunakan**Recharts**(berasaskan SVG) untuk visualisasi analitik interaktif yang boleh diakses (carta bar penggunaan model, jadual pecahan pembekal dengan kadar kejayaan). -7. Ujian E2E menggunakan**Playwright**(`tests/e2e/`), dijalankan melalui `npm run test:e2e`. Ujian unit menggunakan**Node.js test runner**(`tests/unit/`), dijalankan melalui `npm run test:unit`. Kod sumber di bawah `src/` ialah**TypeScript**(`.ts`/`.tsx`); ruang kerja `open-sse/` kekal JavaScript (`.js`). -8. Halaman tetapan disusun dalam 5 tab: Keselamatan, Penghalaan (6 strategi global: isikan dahulu, round-robin, p2c, rawak, paling kurang digunakan, dioptimumkan kos), Ketahanan (had kadar boleh diedit, pemutus litar, dasar), AI (belanjawan berfikir, gesaan sistem, cache segera), Lanjutan (proksi).## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- Bina daripada sumber: `npm run build` -- Bina imej Docker: `docker build -t omniroute .` -- Mulakan perkhidmatan dan sahkan: -- `DAPATKAN /api/tetapan` -- `DAPATKAN /api/v1/model` -- URL asas sasaran CLI hendaklah `http://:20128/v1` apabila `PORT=20128` +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` +- `GET /api/v1/models` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/ms/docs/FEATURES.md b/docs/i18n/ms/docs/FEATURES.md index d5a61a0baf..8d3e271cc7 100644 --- a/docs/i18n/ms/docs/FEATURES.md +++ b/docs/i18n/ms/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Panduan visual untuk setiap bahagian papan pemuka OmniRoute.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Urus sambungan pembekal AI: penyedia OAuth (Claude Code, Codex, Gemini CLI), penyedia kunci API (Groq, DeepSeek, OpenRouter) dan penyedia percuma (Qoder, Qwen, Kiro). Akaun Kiro termasuk penjejakan baki kredit — baki kredit, jumlah elaun dan tarikh pembaharuan boleh dilihat dalam Papan Pemuka → Penggunaan.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Cipta gabungan penghalaan model dengan 6 strategi: keutamaan, wajaran, round-robin, rawak, paling kurang digunakan dan dioptimumkan kos. Setiap kombo merangkai berbilang model dengan sandaran automatik dan termasuk templat pantas dan semakan kesediaan.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Analitis penggunaan komprehensif dengan penggunaan token, anggaran kos, peta haba aktiviti, carta pengedaran mingguan dan pecahan setiap pembekal.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Pemantauan masa nyata: masa aktif, memori, versi, persentil kependaman (p50/p95/p99), statistik cache dan keadaan pemutus litar pembekal.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Empat mod untuk penyahpepijatan terjemahan API:**Taman Permainan**(penukar format),**Penguji Sembang**(permintaan langsung),**Bangku Ujian**(ujian kelompok) dan**Monitor Langsung**(strim masa nyata).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Uji mana-mana model terus dari papan pemuka. Pilih pembekal, model dan titik akhir, tulis gesaan dengan Editor Monaco, strim respons dalam masa nyata, batalkan pertengahan strim dan lihat metrik pemasaan.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Tema warna yang boleh disesuaikan untuk keseluruhan papan pemuka. Pilih daripada 7 warna pratetap (Coral, Biru, Merah, Hijau, Violet, Jingga, Cyan) atau buat tema tersuai dengan memilih mana-mana warna heks. Menyokong mod terang, gelap dan sistem.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Panel tetapan komprehensif dengan tab: +Comprehensive settings panel with tabs: --**Umum**— Storan sistem, pengurusan sandaran (pangkalan data eksport/import) -**Penampilan**— Pemilih tema (gelap/cahaya/sistem), pratetap tema warna dan warna tersuai, keterlihatan log kesihatan, kawalan keterlihatan item bar sisi -**Keselamatan**— Perlindungan titik akhir API, penyekatan pembekal tersuai, penapisan IP, maklumat sesi -**Penghalaan**— Alias model, kemerosotan tugas latar belakang -**Ketahanan**— Kegigihan had kadar, penalaan pemutus litar, nyahdaya automatik akaun terlarang, pemantauan tamat tempoh pembekal -**Lanjutan**— Penggantian konfigurasi, jejak audit konfigurasi, mod degradasi sandaran![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Konfigurasi satu klik untuk alat pengekodan AI: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor dan Factory Droid. Menampilkan penggunaan/set semula konfigurasi automatik, profil sambungan dan pemetaan model.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Papan pemuka untuk menemui dan mengurus ejen CLI. Menunjukkan grid 14 ejen terbina dalam (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) dengan: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Status pemasangan**— Dipasang / Tidak Ditemui dengan pengesanan versi -**Lencana protokol**— stdio, HTTP, dsb. -**Ejen tersuai**— Daftar mana-mana alat CLI melalui borang (nama, binari, arahan versi, pertikaian spawn) -**Padanan Cap Jari CLI**— Togol setiap pembekal untuk memadankan tandatangan permintaan CLI asli, mengurangkan risiko larangan sambil mengekalkan IP proksi--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Hasilkan imej, video dan muzik daripada papan pemuka. Menyokong OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open dan MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Pengelogan permintaan masa nyata dengan penapisan mengikut pembekal, model, akaun dan kunci API. Menunjukkan kod status, penggunaan token, kependaman dan butiran respons.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Titik akhir API bersatu anda dengan pecahan keupayaan: Pelengkapan Sembang, API Respons, Pembenaman, Penjanaan Imej, Kedudukan Semula, Transkripsi Audio, Teks-ke-Pertuturan, Penyederhanaan dan kunci API berdaftar. Penyepaduan Cloudflare Quick Tunnel dan sokongan proksi awan untuk akses jauh.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -Buat, skop dan batalkan kunci API. Setiap kunci boleh dihadkan kepada model/penyedia tertentu dengan akses penuh atau kebenaran baca sahaja. Pengurusan kunci visual dengan penjejakan penggunaan.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Penjejakan tindakan pentadbiran dengan penapisan mengikut jenis tindakan, pelakon, sasaran, alamat IP dan cap masa. Sejarah peristiwa keselamatan penuh.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Apl desktop Native Electron untuk Windows, macOS dan Linux. Jalankan OmniRoute sebagai aplikasi kendiri dengan penyepaduan dulang sistem, sokongan luar talian, kemas kini automatik dan pemasangan satu klik. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -ciri utama: +Key features: -- Undian kesediaan pelayan (tiada skrin kosong pada permulaan sejuk) -- Dulang sistem dengan pengurusan port -- Dasar Keselamatan Kandungan -- Kunci satu contoh -- Kemas kini automatik semasa dimulakan semula -- UI bersyarat platform (lampu isyarat macOS, bar tajuk lalai Windows/Linux) -- Pembungkusan binaan Elektron yang dikeraskan — `node_modules` yang dipautkan dalam himpunan kendiri dikesan dan ditolak sebelum pembungkusan, menghalang pergantungan masa jalan pada mesin binaan (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Lihat [`electron/README.md`](../electron/README.md) untuk dokumentasi penuh. +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/ms/docs/TROUBLESHOOTING.md b/docs/i18n/ms/docs/TROUBLESHOOTING.md index d9b525e5cb..d7e1f9ac41 100644 --- a/docs/i18n/ms/docs/TROUBLESHOOTING.md +++ b/docs/i18n/ms/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -Masalah dan penyelesaian biasa untuk OmniRoute.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Masalah | Penyelesaian | -| ---------------------------------------- | ------------------------------------------------------------------------ | --- | -| Log masuk pertama tidak berfungsi | Tetapkan `INITIAL_PASSWORD` dalam `.env` (tiada lalai berkod keras) | -| Papan pemuka dibuka pada port yang salah | Tetapkan `PORT=20128` dan `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Tiada log permintaan di bawah `log/` | Tetapkan `ENABLE_REQUEST_LOGS=true` | -| EACCES: kebenaran ditolak | Tetapkan `DATA_DIR=/path/to/writable/dir` untuk mengatasi `~/.omniroute` | -| Strategi penghalaan tidak menyimpan | Kemas kini kepada v1.4.11+ (Pembetulan skema Zod untuk tetapan tetapan) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Punca:**Kuota pembekal habis. +**Cause:** Provider quota exhausted. -**Betulkan:** +**Fix:** -1. Semak penjejak kuota papan pemuka -2. Gunakan kombo dengan peringkat sandaran -3. Tukar kepada peringkat yang lebih murah/percuma### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Punca:**Kuota langganan habis. +### Rate Limiting -**Betulkan:** +**Cause:** Subscription quota exhausted. -- Tambahkan sandaran: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Gunakan GLM/MiniMax sebagai sandaran murah### OAuth Token Expired +**Fix:** -Token auto-refresh OmniRoute. Jika isu berterusan: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Papan pemuka → Pembekal → Sambung semula -2. Padam dan tambah semula sambungan pembekal--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Sahkan mata `BASE_URL` pada contoh larian anda (cth., `http://localhost:20128`) -2. Sahkan titik `CLOUD_URL` ke titik akhir awan anda (cth., `https://omniroute.dev`) -3. Pastikan nilai `NEXT_PUBLIC_*` sejajar dengan nilai sebelah pelayan### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Simptom:**`Token 'd' yang tidak dijangka...` pada titik akhir awan untuk panggilan bukan penstriman. +### Cloud `stream=false` Returns 500 -**Punca:**Hulu mengembalikan muatan SSE sementara pelanggan menjangkakan JSON. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Penyelesaian:**Gunakan `strim=true` untuk panggilan terus awan. Masa jalan tempatan termasuk SSE→JSON sandaran.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Cipta kunci baharu daripada papan pemuka setempat (`/api/keys`) -2. Jalankan penyegerakan awan: Dayakan Awan → Segerakkan Sekarang -3. Kekunci lama/tidak disegerakkan masih boleh mengembalikan `401` pada awan--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Semak medan masa jalan: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. Untuk mod mudah alih: gunakan sasaran imej `runner-cli` (CLI yang digabungkan) -3. Untuk mod lekap hos: tetapkan `CLI_EXTRA_PATHS` dan lekapkan direktori bin hos sebagai baca sahaja -4. Jika `installed=true` dan `runnable=false`: binari ditemui tetapi gagal pemeriksaan kesihatan### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Semak statistik penggunaan dalam Papan Pemuka → Penggunaan -2. Tukar model utama kepada GLM/MiniMax -3. Gunakan peringkat percuma (Gemini CLI, Qoder) untuk tugasan yang tidak kritikal -4. Tetapkan belanjawan kos setiap kunci API: Papan Pemuka → Kunci API → Belanjawan--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Tetapkan `ENABLE_REQUEST_LOGS=true` dalam fail `.env` anda. Log muncul di bawah direktori `log/`.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Keadaan utama: `${DATA_DIR}/storage.sqlite` (penyedia, kombo, alias, kunci, tetapan) -- Penggunaan: Jadual SQLite dalam `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + pilihan `${DATA_DIR}/log.txt` dan `${DATA_DIR}/call_logs/` -- Log permintaan: `/logs/...` (apabila `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Apabila pemutus litar pembekal DIBUKA, permintaan disekat sehingga tempoh bertenang tamat. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Betulkan:** +**Fix:** -1. Pergi ke**Papan Pemuka → Tetapan → Ketahanan** -2. Periksa kad pemutus litar untuk pembekal yang terjejas -3. Klik**Tetapkan Semula Semua**untuk mengosongkan semua pemutus, atau tunggu sehingga tempoh bertenang tamat -4. Sahkan pembekal sebenarnya tersedia sebelum menetapkan semula### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Jika pembekal berulang kali memasuki keadaan OPEN: +### Provider keeps tripping the circuit breaker -1. Semak**Papan Pemuka → Kesihatan → Kesihatan Pembekal**untuk corak kegagalan -2. Pergi ke**Tetapan → Ketahanan → Profil Pembekal**dan tingkatkan ambang kegagalan -3. Semak sama ada pembekal telah menukar had API atau memerlukan pengesahan semula -4. Semak telemetri kependaman — kependaman tinggi boleh menyebabkan kegagalan berdasarkan tamat masa--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Pastikan anda menggunakan awalan yang betul: `deepgram/nova-3` atau `assemblyai/best` -- Sahkan pembekal disambungkan dalam**Papan Pemuka → Pembekal**### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Semak format audio yang disokong: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- Sahkan saiz fail berada dalam had pembekal (biasanya < 25MB) -- Semak kesahihan kunci API pembekal dalam kad pembekal--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Gunakan**Papan Pemuka → Penterjemah**untuk menyahpepijat isu terjemahan format: +Use **Dashboard → Translator** to debug format translation issues: -| Mod | Bila Menggunakan | -| --------------------- | ------------------------------------------------------------------------------------------------------------------ | ------------------------ | -| **Taman Permainan** | Bandingkan format input/output sebelah menyebelah — tampal permintaan yang gagal untuk melihat cara ia menterjemah | -| **Penguji Sembang** | Hantar mesej langsung dan periksa muatan penuh permintaan/tindak balas termasuk pengepala | -| **Bangku Ujian** | Jalankan ujian kelompok merentas gabungan format untuk mencari terjemahan yang rosak | -| **Pemantau Langsung** | Tonton aliran permintaan masa nyata untuk menangkap isu terjemahan terputus-putus | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Teg pemikiran tidak muncul**— Semak sama ada pembekal sasaran menyokong pemikiran dan tetapan belanjawan pemikiran -**Panggilan alat terputus**— Sesetengah terjemahan format mungkin menanggalkan medan yang tidak disokong; sahkan dalam mod Taman Permainan -**Gesaan sistem tiada**— Gesaan sistem pengendalian Claude dan Gemini secara berbeza; semak output terjemahan -**SDK mengembalikan rentetan mentah dan bukannya objek**— Ditetapkan dalam v1.1.0: sanitizer respons kini menanggalkan medan bukan standard (`x_groq`, `penggunaan_pecahan`, dsb.) yang menyebabkan kegagalan pengesahan OpenAI SDK Pydantic -**GLM/ERNIE menolak peranan `sistem`**— Ditetapkan dalam v1.1.0: penormal peranan secara automatik menggabungkan mesej sistem ke dalam mesej pengguna untuk model yang tidak serasi -**peranan `pembangun` tidak dikenali**— Ditetapkan dalam v1.1.0: ditukar secara automatik kepada `sistem` untuk penyedia bukan OpenAI -**`json_schema` tidak berfungsi dengan Gemini**— Dibetulkan dalam v1.1.0: `response_format` kini ditukar kepada `responseMimeType` + `responseSchema` Gemini--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- Had kadar automatik hanya digunakan untuk penyedia kunci API (bukan OAuth/langganan) -- Sahkan**Tetapan → Ketahanan → Profil Pembekal**telah didayakan had kadar automatik -- Semak sama ada pembekal mengembalikan kod status `429` atau pengepala `Cuba-Selepas`### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -Profil pembekal menyokong tetapan ini: +### Tuning exponential backoff --**Kelewatan asas**— Masa menunggu awal selepas kegagalan pertama (lalai: 1s) -**Lengah maksimum**— Had masa menunggu maksimum (lalai: 30s) -**Pendarab**— Berapa banyak untuk meningkatkan kelewatan setiap kegagalan berturut-turut (lalai: 2x)### Anti-thundering herd +Provider profiles support these settings: -Apabila banyak permintaan serentak melanda penyedia terhad kadar, OmniRoute menggunakan mutex + pengehadan kadar automatik untuk menyerikan permintaan dan mencegah kegagalan berlatarkan. Ini adalah automatik untuk pembekal kunci API.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Sesetengah pengguna OmniRoute meletakkan get laluan di hadapan RAG atau susunan ejen. Dalam persediaan tersebut adalah perkara biasa untuk melihat corak pelik: OmniRoute kelihatan sihat (penyedia, profil penghalaan ok, tiada makluman had kadar) tetapi jawapan akhir masih salah. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -Dalam praktiknya, kejadian ini biasanya datang dari saluran paip RAG hiliran, bukan dari pintu masuk itu sendiri. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Jika anda mahukan perbendaharaan kata yang dikongsi untuk menerangkan kegagalan tersebut, anda boleh menggunakan WFGY ProblemMap, sumber teks lesen MIT luaran yang mentakrifkan enam belas corak kegagalan RAG / LLM berulang. Pada peringkat tinggi ia meliputi: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- dapatan semula hanyut dan sempadan konteks yang pecah -- indeks kosong atau basi dan kedai vektor -- pembenaman berbanding ketidakpadanan semantik -- isu tetingkap pemasangan dan konteks segera -- logik runtuh dan jawapan terlalu yakin -- rantai panjang dan kegagalan koordinasi ejen -- memori pelbagai ejen dan hanyut peranan -- masalah penempatan dan pesanan bootstrap +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -Ideanya mudah: +The idea is simple: -1. Apabila anda menyiasat respons yang tidak baik, tangkap: - - tugas dan permintaan pengguna - - kombo laluan atau pembekal dalam OmniRoute - - sebarang konteks RAG yang digunakan di hiliran (dokumen yang diambil, panggilan alat, dll) -2. Petakan kejadian kepada satu atau dua nombor Peta Masalah WFGY (`No.1` … `No.16`). -3. Simpan nombor dalam papan pemuka, buku jalanan atau penjejak insiden anda sendiri di sebelah log OmniRoute. -4. Gunakan halaman WFGY yang sepadan untuk memutuskan sama ada anda perlu menukar strategi susunan RAG, retriever atau penghalaan anda. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -Teks penuh dan resipi konkrit hidup di sini (lesen MIT, teks sahaja): +Full text and concrete recipes live here (MIT license, text only): [WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -Anda boleh mengabaikan bahagian ini jika anda tidak menjalankan saluran paip RAG atau ejen di belakang OmniRoute.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**Isu GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Seni Bina**: Lihat [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) untuk mendapatkan butiran dalaman -**Rujukan API**: Lihat [`docs/API_REFERENCE.md`](API_REFERENCE.md) untuk semua titik akhir -**Papan Pemuka Kesihatan**: Semak**Papan Pemuka → Kesihatan**untuk status sistem masa nyata -**Penterjemah**: Gunakan**Papan Pemuka → Penterjemah**untuk menyahpepijat isu format +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/ms/llm.txt b/docs/i18n/ms/llm.txt new file mode 100644 index 0000000000..d27f6864b6 --- /dev/null +++ b/docs/i18n/ms/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Bahasa Melayu) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Gambaran Keseluruhan + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Keselamatan +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/nl/README.md b/docs/i18n/nl/README.md index abfd65aa7b..0bf7fbda5a 100644 --- a/docs/i18n/nl/README.md +++ b/docs/i18n/nl/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Uw universele API-proxy: één eindpunt, meer dan 60 providers, geen downtime. Nu met**MCP-server (25 tools)**,**A2A-protocol**,**Geheugen-/vaardigheidssystemen**en**Electron Desktop-app**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Chat voltooid • Insluitingen • Beeldgeneratie • Video • Muziek • Audio • Herrangschikking •**Webzoeken**• MCP-server • A2A-protocol • 100% TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Uw universele API-proxy: één eindpunt, meer dan 60 providers, geen downtime. [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Website](https://omniroute.online) • [🚀 Snelle start](#-quick-start) • [💡 Kenmerken](#-key-features) • [📖 Documenten](#-documentatie) • [💰 Prijzen](#-prijzen-in-een-oogopslag) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Beschikbaar in:**🇺🇸 [Engels](README.md) | 🇧🇷 [Português (Brazilië)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiaans](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Duits](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyaars](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesië](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenië](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipijns](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -54,552 +61,628 @@ _Uw universele API-proxy: één eindpunt, meer dan 60 providers, geen downtime. ## 📸 Dashboard Preview
-Klik om dashboardscreenshots te bekijken +Click to see dashboard screenshots -| Pagina | Schermafbeelding | -| --------------------- | ------------------------------------------------------ | ---------- | -| **Aanbieders** | ![Aanbieders](docs/screenshots/01-providers.png) | -| **Combo's** | ![Combo's](docs/screenshots/02-combos.png) | -| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | -| **Gezondheid** | ![Gezondheid](docs/screenshots/04-health.png) | -| **Vertaler** | ![Vertaler](docs/screenshots/05-translator.png) | -| **Instellingen** | ![Instellingen](docs/screenshots/06-settings.png) | -| **CLI-hulpmiddelen** | ![CLI-hulpmiddelen](docs/screenshots/07-cli-tools.png) | -| **Gebruikslogboeken** | ![Gebruik](docs/screenshots/08-usage.png) | -| **Eindpunten** | ![Eindpunten](docs/screenshots/09-endpoint.png) |
| +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | + + --- ### 🤖 Free AI Provider for your favorite coding agents -_Verbind elke AI-aangedreven IDE- of CLI-tool via OmniRoute: gratis API-gateway voor onbeperkte codering._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ - + - - - - - - - - - - - +
+ - OpenClaw
- Openklauw + OpenClaw
+ OpenClaw

⭐ 205K
+ NanoBot
NanoBot

- ⭐ 20,9K + ⭐ 20.9K
+ - PicoClaw
- Picoklauw + PicoClaw
+ PicoClaw

- ⭐ 14,6K + ⭐ 14.6K
+ - ZeroClaw
+ ZeroClaw
ZeroClaw

- ⭐ 9,9K + ⭐ 9.9K
+ - IronClaw
- IJzerklauw + IronClaw
+ IronClaw

- ⭐ 2,1K + ⭐ 2.1K
+ - OpenCode
+ OpenCode
OpenCode

⭐ 106K
+ - Codex CLI
- Codex-CLI + Codex CLI
+ Codex CLI

- ⭐ 60,8K + ⭐ 60.8K
+ - Claude Code
+ Claude Code
Claude Code

- ⭐ 67,3K + ⭐ 67.3K
+ Gemini CLI
- Gemini-CLI + Gemini CLI

- ⭐ 94,7K + ⭐ 94.7K
+ - Kilocode
- Kilocode + Kilo Code
+ Kilo Code

- ⭐ 15,5K + ⭐ 15.5K
-📡 Alle agenten maken verbinding via http://localhost:20128/v1 of http://cloud.omniroute.online/v1 — één configuratie, onbeperkte modellen en quota--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Stop met het verspillen van geld en het bereiken van grenzen:** +**Stop wasting money and hitting limits:** -- Het abonnementsquotum verloopt elke maand ongebruikt -- Tarieflimieten voorkomen dat u halverwege codeert -- Dure API's ($20-50/maand per provider) -- Handmatig schakelen tussen providers +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute lost dit op:** +**OmniRoute solves this:** -- ✅**Maximaliseer abonnementen**- Houd quota bij, gebruik elk bit voordat u het opnieuw instelt -- ✅**Automatische terugval**- Abonnement → API-sleutel → Goedkoop → Gratis, geen downtime -- ✅**Multi-account**- Round-robin tussen accounts per provider -- ✅**Universeel**- Werkt met Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, elke CLI-tool--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Word lid van onze community!**[WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Krijg hulp, deel tips en blijf op de hoogte. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Website**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Problemen**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Communitygroep](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Bijdragen**: zie [CONTRIBUTING.md](CONTRIBUTING.md), open een PR of kies een `goed eerste nummer` -**Origineel project**: [9router door decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Wanneer u een probleem opent, voert u de opdracht system-info uit en voegt u het gegenereerde bestand toe:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Dit genereert een `system-info.txt` met uw Node.js-versie, OmniRoute-versie, OS-details, geïnstalleerde CLI-tools (qoder, gemini, claude, codex, antigravity, droid, enz.), Docker/PM2-status en systeempakketten - alles wat we nodig hebben om uw probleem snel te reproduceren. Voeg het bestand rechtstreeks toe aan uw GitHub-probleem.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Elke ontwikkelaar die AI-tools gebruikt, wordt dagelijks met deze problemen geconfronteerd.**OmniRoute is gebouwd om ze allemaal op te lossen: van kostenoverschrijdingen tot regionale blokkades, van kapotte OAuth-stromen tot protocolbewerkingen en bedrijfsobservatie. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability.
-💸 1. "Ik betaal voor een duur abonnement, maar word nog steeds onderbroken door limieten" +💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Ontwikkelaars betalen $20-200/maand voor Claude Pro, Codex Pro of GitHub Copilot. Zelfs als je betaalt, heeft het quotum een ​​plafond: 5 uur gebruik, wekelijkse limieten of tarieflimieten per minuut. Halverwege de codeersessie reageert de provider niet meer en verliest de ontwikkelaar flow en productiviteit. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** --**Smart 4-Tier Fallback**— Als het abonnementsquotum opraakt, wordt automatisch doorgestuurd naar API Key → Goedkoop → Gratis zonder handmatige tussenkomst --**Provider Limits Tracking**— In de cache opgeslagen quota-snapshots worden vernieuwd volgens een schema aan de serverzijde (standaard `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) met handmatige vernieuwing beschikbaar in de gebruikersinterface --**Ondersteuning voor meerdere accounts**— Meerdere accounts per provider met automatische round-robin — als de ene op is, wordt er overgeschakeld naar de volgende --**Aangepaste combo's**— Aanpasbare fallback-ketens met 9 balanceringsstrategieën (prioriteit, gewogen, eerst vullen, round-robin, P2C, willekeurig, minst gebruikt, kostengeoptimaliseerd, strikt willekeurig) --**Codex Business Quota**— Quotabewaking van zakelijke/teamwerkruimte rechtstreeks in het dashboard
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard + +
-🔌 2. "Ik moet meerdere providers gebruiken, maar elk heeft een andere API" +🔌 2. "I need to use multiple providers but each has a different API" -OpenAI gebruikt het ene formaat, Claude (Anthropic) gebruikt een ander, Gemini nog een ander. Als een ontwikkelaar modellen van verschillende providers wil testen of terug wil vallen tussen deze providers, moet hij SDK's opnieuw configureren, eindpunten wijzigen en omgaan met incompatibele formaten. Aangepaste providers (FriendLI, NIM) hebben niet-standaard modeleindpunten. +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** --**Unified Endpoint**— Eén enkele `http://localhost:20128/v1` dient als proxy voor alle 60+ providers --**Formatvertaling**— Automatisch en transparant: OpenAI ↔ Claude ↔ Gemini ↔ Responses API --**Response Sanitization**— Verwijdert niet-standaard velden (`x_groq`, `usage_breakdown`, `service_tier`) die de OpenAI SDK v1.83+ verbreken --**Rolnormalisatie**— Converteert `ontwikkelaar` → `systeem` voor niet-OpenAI-providers; `systeem` → `gebruiker` voor GLM/ERNIE --**Think Tag Extraction**— Extraheert `` blokken uit modellen zoals DeepSeek R1 in gestandaardiseerde `reasoning_content` --**Gestructureerde uitvoer voor Gemini**— `json_schema` → `responseMimeType`/`responseSchema` automatische conversie --**`stream` is standaard ingesteld op `false`**— Sluit aan bij de OpenAI-specificaties en vermijdt onverwachte SSE in Python/Rust/Go SDK's
+- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs + +
-🌐 3. "Mijn AI-provider blokkeert mijn regio/land" +🌐 3. "My AI provider blocks my region/country" -Providers zoals OpenAI/Codex blokkeren de toegang vanuit bepaalde geografische regio's. Gebruikers krijgen fouten zoals `unsupported_country_region_territory` tijdens OAuth- en API-verbindingen. Dit is vooral frustrerend voor ontwikkelaars uit ontwikkelingslanden. +Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** --**Proxyconfiguratie op 3 niveaus**— Configureerbare proxy op 3 niveaus: globaal (al het verkeer), per provider (slechts één provider) en per verbinding/sleutel --**Kleurgecodeerde proxybadges**— Visuele indicatoren: 🟢 globale proxy, 🟡 providerproxy, 🔵verbindingsproxy, waarbij altijd het IP-adres wordt weergegeven --**OAuth-tokenuitwisseling via proxy**— OAuth-stroom gaat ook via de proxy, waardoor `unsupported_country_region_territory` wordt opgelost --**Verbindingstests via proxy**— Verbindingstests gebruiken de geconfigureerde proxy (geen directe bypass meer) --**SOCKS5-ondersteuning**— Volledige SOCKS5-proxyondersteuning voor uitgaande routering --**TLS Fingerprint Spoofing**— Browserachtige TLS-vingerafdruk via `wreq-js` om botdetectie te omzeilen --**🔏 CLI Fingerprint Matching**— Herschikt headers en body-velden zodat ze overeenkomen met de oorspronkelijke binaire CLI-handtekeningen, waardoor het risico op accountmarkeringen drastisch wordt verminderd. Het proxy-IP blijft behouden: u krijgt tegelijkertijd zowel stealth**als**IP-maskering
+- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously + +
-🆓 4. "Ik wil AI gebruiken voor het coderen, maar ik heb geen geld" +🆓 4. "I want to use AI for coding but I have no money" -Niet iedereen kan $ 20-200 per maand betalen voor AI-abonnementen. Studenten, ontwikkelaars uit opkomende landen, hobbyisten en freelancers hebben kosteloos toegang nodig tot kwaliteitsmodellen. +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** --**Free Tier Providers ingebouwd**— Native ondersteuning voor 100% gratis providers: Qoder (5 onbeperkte modellen via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 onbeperkte modellen: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID gratis), Gemini CLI (180K tokens/maand gratis) --**Ollama Cloud**— In de cloud gehoste Ollama-modellen op `api.ollama.com` met gratis laag "Licht gebruik"; gebruik het voorvoegsel `ollamacloud/` --**Alleen gratis combo's**— Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/maand zonder downtime --**NVIDIA NIM Free Access**— ~40 RPM ontwikkelaars-voor altijd gratis toegang tot meer dan 70 modellen op build.nvidia.com (overgang van credits naar pure tarieflimieten) --**Kostengeoptimaliseerde strategie**— Routingstrategie die automatisch de goedkoopste beschikbare provider kiest
+- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider + +
-🔒 5. "Ik moet mijn AI-gateway beschermen tegen ongeautoriseerde toegang" +🔒 5. "I need to protect my AI gateway from unauthorized access" -Bij het blootstellen van een AI-gateway aan het netwerk (LAN, VPS, Docker) kan iedereen met het adres de tokens/quota van de ontwikkelaar gebruiken. Zonder bescherming zijn API's kwetsbaar voor misbruik, snelle injectie en misbruik. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** --**API Key Management**— Generatie, rotatie en bereik per provider met een speciale `/dashboard/api-manager` pagina --**Machtigingen op modelniveau**— Beperk API-sleutels tot specifieke modellen (`openai/*`, jokertekenpatronen), met de schakelaar Alles toestaan/Beperken --**API Endpoint Protection**— Vereist een sleutel voor `/v1/models` en blokkeer specifieke providers uit de lijst --**Auth Guard + CSRF-bescherming**— Alle dashboardroutes beschermd met `withAuth` middleware + CSRF-tokens --**Rate Limiter**— Per-IP-snelheidslimiet met configureerbare vensters --**IP-filtering**— Toelatingslijst/blokkeerlijst voor toegangscontrole --**Prompt Injection Guard**— Sanering tegen kwaadaardige promptpatronen --**AES-256-GCM-codering**— Inloggegevens gecodeerd in rust
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest + +
-🛑 6. "Mijn provider is uitgevallen en ik ben mijn codeerstroom kwijtgeraakt" +🛑 6. "My provider went down and I lost my coding flow" -AI-aanbieders kunnen instabiel worden, 5xx-fouten retourneren of tijdelijke tarieflimieten bereiken. Als een ontwikkelaar afhankelijk is van één enkele provider, worden deze onderbroken. Zonder stroomonderbrekers kunnen herhaalde pogingen de toepassing laten crashen. +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** --**Stroomonderbreker per model**— Automatisch openen/sluiten met configureerbare drempels en cooldown (Gesloten/Open/Half-Open), bereik per model om trapsgewijze blokkades te voorkomen --**Exponentiële uitstel**— Progressieve vertragingen bij nieuwe pogingen --**Anti-Thundering Herd**— Mutex + semafoorbescherming tegen gelijktijdige nieuwe stormen --**Combo Fallback Chains**— Als de primaire provider faalt, valt deze automatisch zonder tussenkomst door de keten --**Combo-stroomonderbreker**— Schakelt falende providers binnen een combo-keten automatisch uit --**Gezondheidsdashboard**— Uptime-monitoring, status van stroomonderbrekers, uitsluitingen, cachestatistieken, p50/p95/p99-latentie
+- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency + +
-🔧 7. "Het configureren van elke AI-tool is vervelend en repetitief" +🔧 7. "Configuring each AI tool is tedious and repetitive" -Ontwikkelaars gebruiken Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Elke tool heeft een andere configuratie nodig (API-eindpunt, sleutel, model). Opnieuw configureren bij het wisselen van provider of model is tijdverspilling. +Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** --**CLI Tools Dashboard**— Speciale pagina met installatie met één klik voor Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline --**GitHub Copilot Config Generator**— Genereert `chatLanguageModels.json` voor VS Code met bulkmodelselectie --**Onboarding Wizard**— Begeleide installatie in 4 stappen voor nieuwe gebruikers --**Eén eindpunt, alle modellen**— Configureer `http://localhost:20128/v1` één keer, krijg toegang tot meer dan 60 providers
+- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers + +
-🔑 8. "Het beheren van OAuth-tokens van meerdere providers is een hel" +🔑 8. "Managing OAuth tokens from multiple providers is hell" -Claude Code, Codex, Gemini CLI, Copilot: ze gebruiken allemaal OAuth 2.0 met aflopende tokens. Ontwikkelaars moeten zich voortdurend opnieuw authenticeren en omgaan met `client_secret ontbreekt`, `redirect_uri_mismatch` en fouten op externe servers. OAuth op LAN/VPS is bijzonder problematisch. +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** --**Automatische tokenvernieuwing**: OAuth-tokens worden op de achtergrond vernieuwd voordat ze verlopen --**OAuth 2.0 (PKCE) ingebouwd**— Automatische stroom voor Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder --**Multi-Account OAuth**— Meerdere accounts per provider via JWT/ID-tokenextractie --**OAuth LAN/Remote Fix**— Privé-IP-detectie voor `redirect_uri` + handmatige URL-modus voor externe servers --**OAuth achter Nginx**— Gebruikt `window.location.origin` voor reverse proxy-compatibiliteit --**Remote OAuth-handleiding**— Stapsgewijze handleiding voor Google Cloud-inloggegevens op VPS/Docker
+- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker + +
-📊 9. "Ik weet niet hoeveel ik uitgeef of waar" +📊 9. "I don't know how much I'm spending or where" -Ontwikkelaars gebruiken meerdere betaalde providers, maar hebben geen uniform beeld van de uitgaven. Elke provider heeft zijn eigen factureringsdashboard, maar er is geen geconsolideerd overzicht. Onverwachte kosten kunnen zich opstapelen. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** --**Cost Analytics Dashboard**— Kostenregistratie per token en budgetbeheer per provider --**Budgetlimieten per niveau**— Uitgavenplafond per niveau dat automatische terugval activeert --**Prijsconfiguratie per model**— Configureerbare prijzen per model --**Gebruiksstatistieken per API-sleutel**— Verzoekaantal en laatst gebruikte tijdstempel per sleutel --**Analytics Dashboard**— Statistiekkaarten, modelgebruiksgrafiek, providertabel met succespercentages en latentie
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency + +
-🐛 10. "Ik kan geen fouten en problemen in AI-oproepen diagnosticeren" +🐛 10. "I can't diagnose errors and problems in AI calls" -Wanneer een oproep mislukt, weet de ontwikkelaar niet of het een snelheidslimiet, een verlopen token, een verkeerd formaat of een providerfout is. Gefragmenteerde logboeken over verschillende terminals. Zonder waarneembaarheid is debuggen een kwestie van vallen en opstaan. +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** --**Unified Logs Dashboard**— 4 tabbladen: aanvraaglogboeken, proxylogboeken, auditlogboeken, console --**Consolelogviewer**— Realtime viewer in terminalstijl met kleurgecodeerde niveaus, automatisch scrollen, zoeken, filteren --**SQLite Proxy Logs**— Persistente logs die het opnieuw opstarten van de server overleven --**Translator Playground**— 4 foutopsporingsmodi: Playground (formaatvertaling), Chat Tester (retour), Testbank (batch), Live Monitor (realtime) --**Request Telemetry**— p50/p95/p99 latentie + X-Request-Id-tracering --**Op bestanden gebaseerd loggen met rotatie**— App-logboeken roteren op basis van grootte, bewaardagen en archiefaantal; oproeplogartefacten roteren op retentiedagen en aantal bestanden --**Systeeminforapport**— `npm run system-info` genereert `system-info.txt` met uw volledige omgeving (knooppuntversie, OmniRoute-versie, besturingssysteem, CLI-tools, Docker/PM2-status). Voeg het toe bij het melden van problemen, zodat u direct kunt triageen.
+- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. + +
-🏗️ 11. "Het implementeren en onderhouden van de gateway is complex" +🏗️ 11. "Deploying and maintaining the gateway is complex" -Het installeren, configureren en onderhouden van een AI-proxy in verschillende omgevingen (lokaal, VPS, Docker, cloud) is arbeidsintensief. Problemen zoals hardgecodeerde paden, 'EACCES' op mappen, poortconflicten en platformonafhankelijke builds zorgen voor extra wrijving. +Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** --**npm globale installatie**— `npm install -g omniroute && omniroute` — klaar --**Docker Multi-Platform**— AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) --**Docker Compose Profiles**— `base` (geen CLI-tools) en `cli` (met Claude Code, Codex, OpenClaw) --**Electron Desktop App**— Native app voor Windows/macOS/Linux met systeemvak, automatisch starten, offlinemodus --**Split-Port-modus**— API en Dashboard op afzonderlijke poorten voor geavanceerde scenario's (reverse proxy, containernetwerken) --**Cloud Sync**— Configureer synchronisatie tussen apparaten via Cloudflare Workers --**DB Backups**— Automatische back-up, herstel, export en import van alle instellingen, met `DISABLE_SQLITE_AUTO_BACKUP` voor extern beheerde back-ups
+- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups + +
-🌍 12. "De interface is alleen in het Engels en mijn team spreekt geen Engels" +🌍 12. "The interface is English-only and my team doesn't speak English" -Teams in niet-Engelssprekende landen, vooral in Latijns-Amerika, Azië en Europa, worstelen met interfaces die alleen in het Engels beschikbaar zijn. Taalbarrières verminderen de adoptie en vergroten de configuratiefouten. +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** --**Dashboard i18n — 30 talen**— Alle 500+ toetsen vertaald, waaronder Arabisch, Bulgaars, Deens, Duits, Spaans, Fins, Frans, Hebreeuws, Hindi, Hongaars, Indonesisch, Italiaans, Japans, Koreaans, Maleis, Nederlands, Noors, Pools, Portugees (PT/BR), Roemeens, Russisch, Slowaaks, Zweeds, Thais, Oekraïens, Vietnamees, Chinees, Filipijns, Engels --**RTL-ondersteuning**— Ondersteuning van rechts naar links voor Arabisch en Hebreeuws --**Meertalige README's**— 30 volledige documentatievertalingen --**Taalkiezer**— Wereldbolpictogram in de koptekst voor realtime schakelen
+- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching + +
-🔄 13. "Ik heb meer nodig dan chatten: ik heb insluitingen, afbeeldingen en audio nodig" +🔄 13. "I need more than chat — I need embeddings, images, audio" -AI is niet alleen het voltooien van chats. Ontwikkelaars moeten afbeeldingen genereren, audio transcriberen, insluitingen voor RAG maken, documenten opnieuw rangschikken en inhoud modereren. Elke API heeft een ander eindpunt en formaat. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** --**Embeddings**— `/v1/embeddings` met 6 providers en 9+ modellen --**Beeldgeneratie**— `/v1/images/generations` met 10 providers en 20+ modellen (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Tekst-naar-video**— `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) en SD WebUI --**Tekst-naar-muziek**— `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) --**Audiotranscriptie**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Tekst-naar-spraak**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + bestaande providers --**Moderaties**— `/v1/moderations` — Veiligheidscontroles van inhoud --**Herrangschikking**— `/v1/rerang` — Herrangschikking van de relevantie van documenten --**Responses API**— Volledige `/v1/responses` ondersteuning voor Codex
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex + +
-🧪 14. "Ik heb geen manier om de kwaliteit van modellen te testen en te vergelijken" +🧪 14. "I have no way to test and compare quality across models" -Ontwikkelaars willen weten welk model het beste is voor hun gebruiksscenario (code, vertaling, redenering), maar handmatig vergelijken gaat traag. Er bestaan ​​geen geïntegreerde evaluatietools. +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** --**LLM-evaluaties**— Golden set-tests met 10 vooraf geladen cases over begroetingen, wiskunde, aardrijkskunde, codegeneratie, JSON-compliance, vertaling, prijsverlaging, veiligheidsweigering --**4 Matchstrategieën**— `exact`, `contains`, `regex`, `custom` (JS-functie) --**Translator Playground Test Bench**— Batchtests met meerdere inputs en verwachte outputs, vergelijking tussen providers --**Chat Tester**— Volledige rondreis met visuele responsweergave --**Live Monitor**— Realtime stream van alle verzoeken die door de proxy stromen
+- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy + +
-📈 15. "Ik moet schalen zonder prestatieverlies" +📈 15. "I need to scale without losing performance" -Naarmate het verzoekvolume groeit, genereren dezelfde vragen dubbele kosten als dezelfde vragen niet in de cache worden opgeslagen. Zonder idempotentie verspillen dubbele aanvragen de verwerking. Tarieflimieten per aanbieder moeten worden gerespecteerd. +As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** --**Semantische cache**— Cache met twee lagen (handtekening + semantisch) verlaagt de kosten en de latentie --**Request Idempotency**— 5s deduplicatievenster voor identieke verzoeken --**Detectie van tarieflimiet**— RPM per provider, minimale tussenruimte en maximale gelijktijdige tracking --**Bewerkbare snelheidslimieten**— Configureerbare standaardinstellingen in Instellingen → Veerkracht met doorzettingsvermogen --**API Key Validation Cache**— 3-tier cache voor productieprestaties --**Gezondheidsdashboard met telemetrie**— p50/p95/p99-latentie, cachestatistieken, uptime
+- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime + +
-🤖 16. "Ik wil modelgedrag wereldwijd controleren" +🤖 16. "I want to control model behavior globally" -Ontwikkelaars die alle antwoorden in een specifieke taal willen, met een specifieke toon, of redeneringstokens willen beperken. Het is onpraktisch om dit in elke tool/verzoek te configureren. +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** --**Systeempromptinjectie**: algemene prompt toegepast op alle verzoeken --**Thinking Budget Validation**— Redenering van tokentoewijzingscontrole per verzoek (passthrough, automatisch, aangepast, adaptief) --**9 Routingstrategieën**— Globale strategieën die bepalen hoe verzoeken worden gedistribueerd --**Wildcard Router**— `provider/*`-patronen routeren dynamisch naar elke provider --**Combo in-/uitschakelen schakelen**— Schakel combo's rechtstreeks vanuit het dashboard in --**Provider wisselen**— Schakel alle verbindingen voor een provider met één klik in/uit --**Geblokkeerde providers**— Sluit specifieke providers uit van de `/v1/models`-lijst
+- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing + +
-🧰 17. "Ik heb MCP-tools nodig als eersteklas productmogelijkheden" +🧰 17. "I need MCP tools as first-class product capabilities" -Veel AI-gateways stellen MCP alleen bloot als een verborgen implementatiedetail. Teams hebben een zichtbare, beheersbare operationele laag nodig. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** -- MCP verschijnt op het dashboardnavigatie- en eindpuntprotocoltabblad -- Speciale MCP-beheerpagina met proces, tools, scopes en audit -- Ingebouwde snelstart voor `omniroute --mcp` en client-onboarding
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding + +
-🧠 18. "Ik heb A2A-orkestratie nodig met synchronisatie- en streamtaakpaden" +🧠 18. "I need A2A orchestration with sync + stream task paths" -Agentworkflows hebben zowel directe antwoorden nodig als langdurige gestreamde uitvoering met levenscycluscontrole. +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** -- A2A JSON-RPC-eindpunt (`POST /a2a`) met `message/send` en `message/stream` -- SSE-streaming met voortplanting van de terminalstatus -- Taaklevenscyclus-API's voor 'tasks/get' en 'tasks/cancel'
+- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` + +
-🛰️ 19. "Ik heb echte MCP-processtatus nodig, geen geraden status" +🛰️ 19. "I need real MCP process health, not guessed status" -Operationele teams moeten weten of MCP daadwerkelijk leeft, en niet alleen of een API bereikbaar is. +Operational teams need to know if MCP is actually alive, not just whether an API is reachable. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** -- Runtime-hartslagbestand met PID, tijdstempels, transport, aantal gereedschappen en scope-modus -- MCP-status-API die hartslag + recente activiteit combineert -- UI-statuskaarten voor proces/uptime/hartslagversheid
+- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness + +
-📋 20. "Ik heb controleerbare MCP-tooluitvoering nodig" +📋 20. "I need auditable MCP tool execution" -Wanneer tools de configuratie muteren of operationele acties activeren, hebben teams forensische traceerbaarheid nodig. +When tools mutate config or trigger ops actions, teams need forensic traceability. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** -- SQLite-ondersteunde auditregistratie voor MCP-toolaanroepen -- Filters op tool, succes/mislukking, API-sleutel en paginering -- Dashboard-audittabel + statistiekeneindpunten voor automatisering
+- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation + +
-🔐 21. "Ik heb specifieke MCP-rechten nodig per integratie" +🔐 21. "I need scoped MCP permissions per integration" -Verschillende clients moeten toegang tot de toolcategorieën met de minste bevoegdheden hebben. +Different clients should have least-privilege access to tool categories. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** -- 10 gedetailleerde MCP-scopes voor gecontroleerde toegang tot tools -- Bereikafdwinging en zichtbaarheid in de gebruikersinterface voor MCP-beheer -- Veilige standaardhouding voor operationeel gereedschap
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling + +
-⚙️ 22. "Ik heb operationele controles nodig zonder opnieuw te implementeren" +⚙️ 22. "I need operational controls without redeploying" -Teams hebben snelle runtimewijzigingen nodig tijdens incidenten of kostengebeurtenissen. +Teams need quick runtime changes during incidents or cost events. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** -- Schakel combo-activering rechtstreeks vanuit het MCP-dashboard -- Pas veerkrachtprofielen toe uit vooraf gedefinieerde beleidspakketten -- Reset de status van de stroomonderbreker vanaf hetzelfde bedieningspaneel
+- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel + +
-🔄 23. "Ik heb live zichtbaarheid en annulering van de levenscyclus van A2A-taken nodig" +🔄 23. "I need live A2A task lifecycle visibility and cancellation" -Zonder inzicht in de levenscyclus worden taakincidenten moeilijk te beoordelen. +Without lifecycle visibility, task incidents become hard to triage. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** -- Takenlijst/filteren op staat/vaardigheid met paginering -- Inzoomen op taakmetagegevens, gebeurtenissen en artefacten -- Eindpunt voor het annuleren van taken en UI-actie met bevestiging
+- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation + +
-🌊 24. "Ik heb actieve streamstatistieken nodig voor A2A-belasting" +🌊 24. "I need active stream metrics for A2A load" -Streamingworkflows vereisen operationeel inzicht in gelijktijdigheid en liveverbindingen. +Streaming workflows require operational insight into concurrency and live connections. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** -- Actieve streamtellers geïntegreerd in de A2A-status -- Tijdstempel van de laatste taak en tellingen per staat -- A2A-dashboardkaarten voor real-time operationele monitoring
+- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring + +
-🪪 25. "Ik heb standaard agentdetectie nodig voor klanten" +🪪 25. "I need standard agent discovery for clients" -Externe klanten en orkestrators hebben machinaal leesbare metagegevens nodig voor onboarding. +External clients and orchestrators need machine-readable metadata for onboarding. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** -- Agentkaart weergegeven op `/.well-known/agent.json` -- Mogelijkheden en vaardigheden weergegeven in de management-UI -- A2A-status-API bevat ontdekkingsmetagegevens voor automatisering
+- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
-🧭 26. "Ik heb protocolvindbaarheid nodig in de product-UX" +🧭 26. "I need protocol discoverability in the product UX" -Als gebruikers protocoloppervlakken niet kunnen ontdekken, neemt de acceptatie- en ondersteuningskwaliteit af. +If users cannot discover protocol surfaces, adoption and support quality drop. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** -- Geconsolideerde pagina**Eindpunten**met tabbladen voor proxy-, MCP-, A2A- en API-eindpunten -- Inline servicestatus schakelt (online/offline) voor MCP en A2A -- Links van overzicht naar speciale beheertabbladen
+- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
-🧪 27. "Ik heb end-to-end protocolvalidatie nodig met echte clients" +🧪 27. "I need end-to-end protocol validation with real clients" -Mock-tests zijn niet voldoende om de protocolcompatibiliteit vóór de release te valideren. +Mock tests are not enough to validate protocol compatibility before release. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** -- E2E-suite die de app opstart en echt MCP SDK-clienttransport gebruikt -- A2A-clienttests voor het ontdekken, verzenden, streamen, ophalen en annuleren van stromen -- Controleer beweringen aan de hand van MCP-audit- en A2A-taken-API's
+- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
-📡 28. "Ik heb uniforme observatie nodig over alle interfaces heen" +📡 28. "I need unified observability across all interfaces" -Het opsplitsen van de waarneembaarheid per protocol creëert blinde vlekken en een langere MTTR. +Splitting observability by protocol creates blind spots and longer MTTR. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** -- Uniforme dashboards/logboeken/analyses in één product -- Gezondheid + audit + verzoektelemetrie over OpenAI-, MCP- en A2A-lagen -- Operationele API's voor status en automatisering
+- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
-💼 29. "Ik heb één runtime nodig voor proxy + tools + agentorkestratie" +💼 29. "I need one runtime for proxy + tools + agent orchestration" -Het uitvoeren van veel afzonderlijke services verhoogt de operationele kosten en faalwijzen. +Running many separate services increases operational cost and failure modes. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** -- OpenAI-compatibele proxy, MCP-server en A2A-server in één stapel -- Gedeelde authenticatie, veerkracht, gegevensopslag en waarneembaarheid -- Consistent beleidsmodel op alle interactieoppervlakken
+- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
-🚀 30. "Ik moet agentische workflows verzenden zonder wildgroei van lijmcodes" +🚀 30. "I need to ship agentic workflows without glue-code sprawl" -Teams verliezen snelheid bij het samenvoegen van meerdere ad-hocservices en scripts. +Teams lose velocity when stitching multiple ad-hoc services and scripts. -**Hoe OmniRoute het oplost:** +**How OmniRoute solves it:** -- Uniforme eindpuntstrategie voor klanten en agenten -- Ingebouwde gebruikersinterfaces voor protocolbeheer en rookvalidatiepaden -- Productieklare fundamenten (beveiliging, loggen, veerkracht, back-up)
+- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + + ### Example Playbooks (Integrated Use Cases) -**Playbook A: Maximaliseer betaald abonnement + goedkope back-up**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -607,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Playbook B: Codeerstapel zonder kosten**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Playbook C: 24/7, altijd actieve fallback-keten**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -630,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Playbook D: Agentoperaties met MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Stel AI-codering in enkele minuten in voor**$0/maand**. Verbind deze gratis accounts en gebruik de ingebouwde**Gratis Stack**-combinatie. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Stap | Actie | Aanbieders ontgrendeld | -| ---- | --------------------------------------------- | ---------------------------------------------------------- | -| 1 | Verbind**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**onbeperkt**| -| 2 |**Qoder**(Google OAuth) verbinden | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... —**onbeperkt**| -| 3 | Verbind**Qwen**(apparaatcode) | qwen3-coder-plus, qwen3-coder-flash... —**onbeperkt**| -| 4 |**Gemini CLI**verbinden (Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180K/maand gratis**| -| 5 | `/dashboard/combos` →**Gratis stapel ($0)**sjabloon | Round-robin alle gratis aanbieders automatisch | +| Step | Action | Providers Unlocked | +| ---- | -------------------------------------------------- | ------------------------------------------------------------------ | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Wijs een IDE/CLI naar:**`http://localhost:20128/v1` · API-sleutel: `any-string` · Klaar. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Optionele extra dekking (ook gratis):**Groq API-sleutel (30 RPM gratis), NVIDIA NIM (40 RPM gratis, 70+ modellen), Cerebras (1M tok/dag), LongCat API-sleutel (50M tokens/dag!), Cloudflare Workers AI (10K Neurons/dag, 50+ modellen).## Snel starten +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Snel starten ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **pnpm-gebruikers:**Voer `pnpm Approved-builds -g` uit na de installatie om native build-scripts in te schakelen die vereist zijn door `better-sqlite3` en `@swc/core`: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > > ```bash > pnpm install -g omniroute -> pnpm goedkeuring-builds -g # Selecteer alle pakketten → goedkeuren +> pnpm approve-builds -g # Select all packages → approve > omniroute > ``` -Dashboard wordt geopend op `http://localhost:20128` en de API-basis-URL is `http://localhost:20128/v1`. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Commando | Beschrijving | -| ------------------------ | -------------------------------------------------------------- | -| `omniroute` | Startserver (`PORT=20128`, API en dashboard op dezelfde poort) | -| `omniroute --poort 3000` | Stel de canonieke/API-poort in op 3000 | -| `omniroute --mcp` | MCP-server starten (stdio-transport) | -| `omniroute --no-open` | Browser niet automatisch openen | -| `omniroute --help` | Hulp tonen | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Optionele split-port-modus:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -Voor de meeste implementaties heeft u alleen het volgende nodig: +For most deployments, you only need: -| Variabel | Standaard | Doel | -| ----------------------- | --------------------------- | ------------------------------------------------------- ------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600000` | Gedeelde basislijn voor upstream ophalen, verborgen Undici-time-outs, TLS-vingerafdrukverzoeken en API-bridgeverzoeken/proxy-time-outs | -| `STREAM_IDLE_TIMEOUT_MS` | erft `REQUEST_TIMEOUT_MS` | Maximale afstand tussen streaming-brokken voordat OmniRoute de SSE-stream afbreekt | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -Achterwaartse compatibiliteit blijft behouden: bestaande `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` en andere time-outvariabelen per laag werken nog steeds en overschrijven de gedeelde basislijn. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Er zijn geavanceerde overschrijvingen beschikbaar als u een nauwkeurigere controle nodig heeft:| Variabel | Standaard | Doel | -| ------------------------------------- | --------------------------------------- | ------------------------------------------------------------ | -| `FETCH_TIMEOUT_MS` | erft `REQUEST_TIMEOUT_MS` | Totale time-out van het upstream-verzoek gebruikt door het signaal voor het afbreken van het hoofdophalen | -| `FETCH_HEADERS_TIMEOUT_MS` | erft `FETCH_TIMEOUT_MS` | Undici-tijdslimiet voor het ontvangen van upstream-antwoordheaders | -| `FETCH_BODY_TIMEOUT_MS` | erft `FETCH_TIMEOUT_MS` | Undici tijdslimiet tussen upstream body chunks (`0` schakelt dit uit) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP-verbindingstime-out | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici inactieve keep-alive socket time-out | -| `TLS_CLIENT_TIMEOUT_MS` | erft `FETCH_TIMEOUT_MS` | Time-out voor TLS-vingerafdrukverzoeken gedaan via `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | erft `REQUEST_TIMEOUT_MS` of `30000` | Time-out voor het doorsturen van `/v1` proxy van API-poort naar dashboardpoort | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Time-out voor binnenkomend verzoek op de API-bridgeserver | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Time-out voor inkomende header op de API-bridgeserver | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Time-out voor keep-alive op de API-bridgeserver | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Time-out voor socket-inactiviteit op de API-bridgeserver (`0` schakelt dit uit) | +Advanced overrides are available if you need finer control: -Als u OmniRoute achter Nginx, Caddy, Cloudflare of een andere reverse proxy uitvoert, zorg er dan voor dat de proxy -time-outs zijn ook hoger dan uw OmniRoute-time-outs voor streamen/ophalen.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Open Dashboard → `Providers` en verbind minimaal één provider (OAuth of API-sleutel). -2. Open Dashboard → `Eindpunten` en maak een API-sleutel aan. -3. (Optioneel) Open Dashboard → `Combo's` en stel uw fallback-keten in.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Werkt met Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode en OpenAI-compatibele SDK's.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (voor gereedschapgestuurde bewerkingen):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -Verbind vervolgens uw MCP-client via `stdio` en test tools zoals: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` -- `omniroute_lijst_combo`s` +- `omniroute_list_combos` -**A2A (voor agent-tot-agent-workflows):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -759,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Deze suite valideert echte MCP- en A2A-clientstromen tegen een actieve app.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -768,12 +867,12 @@ PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm ```
-Linux ongeldig maken (`xbps-src` sjabloon) +Void Linux (`xbps-src` template) -Voor Void Linux-gebruikers kun je een native pakket bouwen met `xbps-src`. Sla dit blok op als `srcpkgs/omniroute/template`:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -785,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -793,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -869,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -880,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute is beschikbaar als openbare Docker-image op [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Snelle uitvoering:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -890,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**Met omgevingsbestand:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Met behulp van Docker Compose:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Dashboardondersteuning voor Docker-implementaties omvat nu een**Cloudflare Quick Tunnel**met één klik op `Dashboard → Endpoints`. De eerste optie downloadt `cloudflared` alleen wanneer dat nodig is, start een tijdelijke tunnel naar uw huidige `/v1` eindpunt en toont de gegenereerde `https://*.trycloudflare.com/v1` URL direct onder uw normale openbare URL. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Opmerkingen: +Notes: -- Quick Tunnel-URL's zijn tijdelijk en veranderen na elke herstart. -- Quick Tunnels worden niet automatisch hersteld na een herstart van OmniRoute of een container. Schakel ze indien nodig opnieuw in vanaf het dashboard. -- Beheerde installatie ondersteunt momenteel Linux, macOS en Windows op `x64` / `arm64`. -- Beheerde Quick Tunnels gebruiken standaard HTTP/2-transport om luidruchtige QUIC UDP-bufferwaarschuwingen in beperkte containeromgevingen te voorkomen. Stel `CLOUDFLARED_PROTOCOL=quic` of `auto` in als u een ander transport wilt. -- Docker-images bundelen de CA-roots van het systeem en geven deze door aan het beheerde `cloudflared`, waardoor TLS-vertrouwensfouten worden vermeden wanneer de tunnel in de container opstart. -- SQLite draait in WAL-modus. `docker stop` moet worden voltooid, zodat OmniRoute de laatste wijzigingen kan terugzetten in `storage.sqlite`. -- De gebundelde Compose-bestanden stellen al een stopperiode van 40 seconden in. Als u de image rechtstreeks uitvoert, houd dan `--stop-timeout 40` (of iets dergelijks) aan, zodat handmatig stoppen het opschonen van het afsluiten niet onderbreekt. -- Stel `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` in als u wilt dat OmniRoute een bestaand binair bestand gebruikt in plaats van er een te downloaden. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Docker Compose gebruiken met Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute kan veilig worden blootgesteld met behulp van de automatische SSL-voorziening van Caddy. Zorg ervoor dat de DNS A-record van uw domein verwijst naar het IP-adres van uw server.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` +| Image | Tag | Size | Description | +| ------------------------ | -------- | ------ | --------------------- | +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | -| Afbeelding | Label | Maat | Beschrijving | -| ----------------------- | -------- | ------ | -------------------- | -| `diegosouzapw/omniroute` | `nieuwste` | ~250MB | Nieuwste stabiele release | -| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Huidige versie |--- +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**NIEUW!**OmniRoute is nu beschikbaar als**native desktop-applicatie**voor Windows, macOS en Linux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Voer OmniRoute uit als een zelfstandige desktop-app: geen terminal, geen browser, geen internet vereist voor lokale modellen. De op Electron gebaseerde app omvat: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Native Window**— Speciaal app-venster met systeemvakintegratie -- 🔄**Autostart**— Start OmniRoute bij systeemaanmelding -- 🔔**Native Notificaties**— Ontvang waarschuwingen voor opgebruikte quota of problemen met providers -- ⚡**Installeren met één klik**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Offlinemodus**— Werkt volledig offline met gebundelde server### Snel starten +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Snel starten ```bash # Development mode @@ -979,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -Wanneer geminimaliseerd, blijft OmniRoute in uw systeemvak staan met snelle acties: +When minimized, OmniRoute lives in your system tray with quick actions: -- Dashboard openen -- Wijzig de serverpoort -- Sluit de applicatie af +- Open dashboard +- Change server port +- Quit application -📖 Volledige documentatie: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Niveau | Aanbieder | Kosten | Quotum opnieuw instellen | Beste voor | -| ------------------ | --------------------------- | ------------------------------------ | ------------------------ | --------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 ABONNEMENT** | Claude Code (Pro) | $ 20/maand | 5u + wekelijks | Al geabonneerd | -| | Codex (Plus/Pro) | $ 20-200/maand | 5u + wekelijks | OpenAI-gebruikers | -| | Tweeling CLI | **GRATIS** | 180K/maand + 1K/dag | Iedereen! | -| | GitHub-copiloot | $ 10-19/maand | Maandelijks | GitHub-gebruikers | -| **🔑 API-SLEUTEL** | NVIDIA NIM | **GRATIS**(ontwikkelaar voor altijd) | ~40 tpm | 70+ open modellen | -| | Hersenen | **GRATIS**(1 miljoen tok/dag) | 60K TPM / 30 RPM | 's Werelds snelste | -| | Groq | **GRATIS**(30 toeren) | 14,4K RPD | Ultrasnelle lama/gemma | -| | DeepSeek V3.2 | $0,27/$1,10 per 1 miljoen | Geen | Beste prijs/kwaliteit-redenering | -| | xAI Grok-4 Snel | **$0,20/$0,50 per 1M**🆕 | Geen | Snelste + gereedschapsoproep, ultralaag | -| | xAI Grok-4 (standaard) | $ 0,20/$ 1,50 per 1 miljoen 🆕 | Geen | Redenerend vlaggenschip van xAI | -| | Mistral | Gratis proefversie + betaald | Tarief beperkt | Europese AI | -| | OpenRouter | Betalen per gebruik | Geen | 100+ modellen aggr. | -| **💰GOEDKOOP** | GLM-5 (via Z.AI) 🆕 | $ 0,5/1 miljoen | Dagelijks 10.00 uur | 128K-uitvoer, nieuwste vlaggenschip | -| | GLM-4.7 | $ 0,6/1 miljoen | Dagelijks 10.00 uur | Budgetback-up | -| | MiniMax M2.5 🆕 | $0,3/1 miljoen invoer | 5-uurs rollen | Redeneren + agenttaken | -| | MiniMax M2.1 | $ 0,2/1 miljoen | 5-uurs rollen | Goedkoopste optie | -| | Kimi K2.5 (Moonshot-API) 🆕 | Betalen per gebruik | Geen | Directe Moonshot API-toegang | -| | Kimi K2 | $ 9/maand plat | 10 miljoen tokens/maand | Voorspelbare kosten | -| **🆓 GRATIS** | Qoder | **$0** | Onbeperkt | 5 modellen onbeperkt | -| | Qwen | **$0** | Onbeperkt | 4 modellen onbeperkt | -| | Kiro | **$0** | Onbeperkt | Claude Sonnet/Haiku (AWS-bouwer) | -| | LongCat Flash-Lite 🆕 | **$0**(50 miljoen tok/dag 🔥) | 1 RPS | Grootste gratis quotum ter wereld | -| | Bestuivingen AI 🆕 | **$0**(geen sleutel nodig) | 1 verzoek/15s | GPT-5, Claude, DeepSeek, Lama 4 | -| | Cloudflare Workers AI 🆕 | **$0**(10.000 neuronen/dag) | ~150 resp./dag | 50+ modellen, mondiale voorsprong | -| | Scaleway-AI 🆕 | **$0**(totaal 1 miljoen tokens) | Tarief beperkt | EU/AVG, Qwen3 235B, Lama 70B | > 🆕**Nieuwe modellen toegevoegd (maart 2026):**Grok-4 Fast-familie voor $ 0,20/$ 0,50/M (benchmarked op 1143 ms - 30% sneller dan Gemini 2.5 Flash), GLM-5 via Z.AI met 128K-uitvoer, MiniMax M2.5-redenering, DeepSeek V3.2 bijgewerkte prijzen, Kimi K2.5 via Moonshot direct API. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 $0 Combo Stack — De complete gratis installatie:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Nul kosten. Stopt nooit met coderen.**Configureer dit als één OmniRoute-combo en alle fallbacks gebeuren automatisch - nooit handmatig schakelen.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Alle onderstaande modellen zijn**100% gratis, zonder creditcard**. OmniRoute berekent automatisch routes ertussen wanneer één quotum op is – combineer ze allemaal voor een onbreekbare combinatie van $ 0.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Model | Voorvoegsel | Limiet | Tarieflimiet | -| ------------------ | ------ | ------------- | -------------------- | -| `claude-sonnet-4.5` | `kr/` |**Onbeperkt**| Geen gerapporteerde dagelijkse limiet | -| `claude-haiku-4.5` | `kr/` |**Onbeperkt**| Geen gerapporteerde dagelijkse limiet | -| `claude-opus-4.6` | `kr/` |**Onbeperkt**| Nieuwste opus via Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) -| Model | Voorvoegsel | Limiet | Tarieflimiet | +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | --------------------- | +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | + +### 🟢 QODER MODELS (Free PAT via qodercli) + +| Model | Prefix | Limit | Rate Limit | | ------------------ | ------ | ------------- | --------------- | -| `kimi-k2-denken` | `als/` |**Onbeperkt**| Geen gerapporteerde limiet | -| `qwen3-coder-plus` | `als/` |**Onbeperkt**| Geen gerapporteerde limiet | -| `diepzoek-r1` | `als/` |**Onbeperkt**| Geen gerapporteerde limiet | -| `minimax-m2.1` | `als/` |**Onbeperkt**| Geen gerapporteerde limiet | -| `kimi-k2` | `als/` |**Onbeperkt**| Geen gerapporteerde limiet | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -> Aanbevolen verbindingsmethode:**Persoonlijk toegangstoken + `qodercli`**. Browser-OAuth is -> experimenteel en standaard uitgeschakeld tenzij `QODER_OAUTH_*` omgevingsvariabelen zijn geconfigureerd.### 🟡 QWEN MODELS (Device Code Auth) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| Model | Voorvoegsel | Limiet | Tarieflimiet | -| ------------------ | ------ | ------------- | ------------------ | -| `qwen3-coder-plus` | `qw/` |**Onbeperkt**| Geen gerapporteerde limiet | -| `qwen3-coder-flash` | `qw/` |**Onbeperkt**| Geen gerapporteerde limiet | -| `qwen3-coder-volgende` | `qw/` |**Onbeperkt**| Geen gerapporteerde limiet | -| `visiemodel` | `qw/` |**Onbeperkt**| Multimodaal (afbeeldingen) |### 🟣 GEMINI CLI (Google OAuth) +### 🟡 QWEN MODELS (Device Code Auth) -| Model | Voorvoegsel | Limiet | Tarieflimiet | -| ----------------------- | ------ | ------------------------- | ------------- | -| `gemini-3-flash-preview` | `gc/` |**180K tok/maand**+ 1K/dag | Maandelijkse reset | -| `gemini-2.5-pro` | `gc/` | 180K/maand (gedeeld zwembad) | Hoge kwaliteit |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | ------------------- | +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Niveau | Dagelijkse limiet | Tarieflimiet | Opmerkingen | -| ---------- | ------------ | ----------- | ------------------------------------------------- | -| Gratis (ontwikkelaar) | Geen tokenlimiet |**~40 tpm**| 70+ modellen; overgang naar zuivere tarieflimieten medio 2025 | +### 🟣 GEMINI CLI (Google OAuth) -Populaire gratis modellen: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -| Niveau | Dagelijkse limiet | Tarieflimiet | Opmerkingen | -| ---- | ----------------- | ---------------- | --------------------------------------- | -| Gratis |**1 miljoen tokens/dag**| 60K TPM / 30 RPM | 's Werelds snelste LLM-gevolgtrekking; wordt dagelijks gereset | +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) -Gratis verkrijgbaar: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +| Tier | Daily Limit | Rate Limit | Notes | +| ---------- | ------------ | ----------- | ------------------------------------------------------ | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -| Niveau | Dagelijkse limiet | Tarieflimiet | Opmerkingen | -| ---- | ------------- | ---------------- | -------------------------------------- | -| Gratis |**14,4K RPD**| 30 RPM per model | Geen creditcard; 429 op limiet, niet in rekening gebracht | +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -Gratis verkrijgbaar: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -| Model | Voorvoegsel | Dagelijks gratis quotum | Opmerkingen | -| --------------------------- | ------ | ----------------- | ---------------------- | -| `LongCat-Flash-Lite` | `lc/` |**50 miljoen tokens**💥 | Grootste gratis quotum ooit | -| `LongCat-Flash-Chat` | `lc/` | 500K-tokens | Chat met meerdere beurten | -| `LongCat-Flash-denken` | `lc/` | 500K-tokens | Redenering / CoT | -| `LongCat-Flash-Thinking-2601` | `lc/` | 500K-tokens | Versie januari 2026 | -| `LongCat-Flash-Omni-2603` | `lc/` | 500K-tokens | Multimodaal | +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -> 100% gratis tijdens de openbare bèta. Meld u aan op [longcat.chat](https://longcat.chat) met e-mail of telefoon. Wordt dagelijks gereset om 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -| Model | Voorvoegsel | Tarieflimiet | Aanbieder achter | +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ------------- | ---------------- | ----------------------------------------- | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | + +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` + +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 + +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | + +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. + +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| `openai` | `pol/` | 1 verzoek/15s | GPT-5 | -| `claude` | `pol/` | 1 verzoek/15s | Antropische Claude | -| `Tweeling` | `pol/` | 1 verzoek/15s | Google Tweeling | -| `diep zoeken` | `pol/` | 1 verzoek/15s | DeepSeek V3 | -| `lama` | `pol/` | 1 verzoek/15s | Meta Lama 4 Scout | -| `mistral` | `pol/` | 1 verzoek/15s | Mistral-AI | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**Geen wrijving:**Geen aanmelding, geen API-sleutel. Voeg de Pollinations-provider toe met een leeg sleutelveld en het werkt meteen.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| Niveau | Dagelijkse neuronen | Equivalent gebruik | Opmerkingen | -| ---- | ------------- | ------------------------------------ | ---------------------- | -| Gratis |**10.000**| ~150 LLM resp / 500s audio / 15K insluitingen | Wereldwijde voorsprong, 50+ modellen | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 -Populaire gratis modellen: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (gratis audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` +| Tier | Daily Neurons | Equivalent Usage | Notes | +| ---- | ------------- | --------------------------------------- | ----------------------- | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -> Vereist API-token + account-ID van [dash.cloudflare.com](https://dash.cloudflare.com). Bewaar account-ID in providerinstellingen.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -| Niveau | Gratis quotum | Locatie | Opmerkingen | -| ---- | ------------- | ------------ | ---------------------------------- | -| Gratis |**1 miljoen tokens**| 🇫🇷 Parijs, EU | Binnen bepaalde grenzen geen creditcard nodig | +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -Gratis verkrijgbaar: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 -> EU/AVG-compatibel. Haal de API-sleutel op op [console.scaleway.com](https://console.scaleway.com). +| Tier | Free Quota | Location | Notes | +| ---- | ------------- | ------------ | ----------------------------------- | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | ->**💡 De ultieme gratis stapel (11 providers, $ 0 voor altijd):** +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` + +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). + +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Sonnet/Haiku ONBEPERKT -> Qoder (if/) → kimi-k2-denken, qwen3-coder-plus, deepseek-r1 ONBEPERKT -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 miljoen tokens/dag 🔥 -> Bestuivingen (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — geen sleutel nodig -> Qwen (qw/) → qwen3-codermodellen ONBEPERKT -> Gemini (gemini/) → Gemini 2.5 Flash — 1.500 vereist/dag gratis -> Cloudflare AI (cf/) → 50+ modellen — 10K neuronen/dag -> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M gratis tokens (EU) -> Groq (groq/) → Llama/Gemma — 14,4K vereist/dag ultrasnel -> NVIDIA NIM (nvidia/) → 70+ open modellen — altijd 40 RPM -> Cerebras (cerebras/) → Lama/Qwen snelste ter wereld — 1M tok/dag -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Transcribeer audio/video voor**$0**— Deepgram-leads met $200 gratis, AssemblyAI $50 fallback, Groq Whisper als onbeperkte noodback-up. +## 🎙️ Free Transcription Combo -| Aanbieder | Gratis tegoeden | Beste model | Tarieflimiet | -| ----------------- | ---------------------- | ---------------------------------------- | -------------------------- | -| 🟢**Deepgram**|**$200 gratis**(aanmelding) | `nova-3` — beste nauwkeurigheid, meer dan 30 talen | Geen RPM-limiet op gratis credits | -| 🔵**AssemblageAI**|**$ 50 gratis**(aanmelding) | `universal-3-pro` — hoofdstukken, sentiment, PII | Geen RPM-limiet op gratis credits | -| 🔴**Groq**|**Voor altijd gratis**| `fluister-groot-v3` — OpenAI Whisper | 30 RPM (snelheid beperkt) | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. -**Voorgestelde combo in `/dashboard/combos`:**``` +| Provider | Free Credits | Best Model | Rate Limit | +| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | + +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Vervolgens in `/dashboard/media` → tabblad**Transcriptie**: upload een audio- of videobestand → selecteer uw combo-eindpunt → ontvang transcriptie in ondersteunde formaten.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 is gebouwd als een operationeel platform, niet alleen als een relay-proxy.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Kenmerk | Wat het doet | -| --------------------------------- | ---------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Grok-4 Fast Familie** | xAI-modellen voor $ 0,20/$ 0,50/M — gebenchmarkt 1143 ms (30% sneller dan Gemini 2.5 Flash) | -| 🧠**GLM-5 via Z.AI** | 128K outputcontext, $0,5/1M — nieuwste vlaggenschip uit de GLM-familie | -| 🔮**MiniMax M2.5** | Redeneren + agenttaken voor $ 0,30/1 miljoen – aanzienlijke upgrade van M2.1 | -| 🎯**toolCalling-vlag per model** | Per model `toolCalling: true/false` in register: AutoCombo slaat modellen die niet geschikt zijn voor tools over | -| 🌍**Meertalige intentiedetectie** | PT/ZH/ES/AR-trefwoorden in AutoCombo-scores — betere modelselectie voor niet-Engelse inhoud | -| 📊**Benchmarkgestuurde terugval** | Echte p95-latentie van live-verzoeken feeds combo-scores - AutoCombo leert van feitelijke gegevens | -| 🔁**Ontdubbeling aanvragen** | Op inhoud-hash gebaseerd ontdubbelingsvenster — veilig voor meerdere agenten, voorkomt dubbele kosten | -| 🔌**Inplugbare routerstrategie** | Uitbreidbare `RouterStrategy`-interface — voeg aangepaste routeringslogica toe als plug-ins | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Kenmerk | Wat het doet | -| ----------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Modelspeeltuin** | Dashboardpagina om elk model rechtstreeks te testen: provider/model/eindpuntkiezers, Monaco Editor, streaming, afbreken, timing | -| 🔏**CLI-vingerafdrukmatching** | Kop-/tekstvolgorde per provider zodat deze overeenkomt met de oorspronkelijke CLI-handtekeningen: schakel per provider in Instellingen > Beveiliging.**Uw proxy-IP blijft behouden** | -| 🤝**ACP-ondersteuning (Agent Client Protocol)** | CLI-agentdetectie (Codex, Claude, Goose, Gemini CLI, OpenClaw + nog 9), process spawner, `/api/acp/agents` eindpunt | -| 🤖**ACP-agentendashboard** | Debug › Agentenpagina — raster van 14 agenten met installatiestatus, versie en aangepast agentformulier voor elke CLI-tool.**OpenCode**-gebruikers krijgen een knop 'Opencode.json downloaden' die automatisch een gebruiksklare configuratie genereert met alle beschikbare modellen. | -| 🔧**Aangepast model `apiFormat` Routing** | Aangepaste modellen met `apiFormat: "responses"` routeren nu correct naar de Responses API-vertaler | -| 🏢**Codex-werkruimte-isolatie** | Meerdere Codex-werkruimten per e-mail: OAuth scheidt verbindingen correct op werkruimte-ID | -| 🔄**Elektronische automatische update** | Desktop-app controleert op updates + automatisch installeren bij opnieuw opstarten | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Kenmerk | Wat het doet | -| --------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**MCP-server (25 tools)** | IDE/agent-tools via 3 transporten: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 geheugen + 4 vaardigheidstools | -| 🤝**A2A-server (JSON-RPC + SSE)** | Agent-naar-agent-taakuitvoering met synchronisatie- en streamingstromen | -| 🧭**Geconsolideerde eindpuntenpagina** | Beheerpagina met tabbladen met tabbladen Endpoint Proxy, MCP, A2A en API Endpoints | -| 🎚️**Service in-/uitschakelen schakelt** | AAN/UIT-schakelaars voor MCP en A2A met persistentie van de instellingen (standaard: UIT) | -| 🛰️**MCP Runtime-hartslag** | Echte processtatus (pid, uptime, hartslagleeftijd, transport, scopemodus) | -| 📋**MCP-audittraject** | Filterbare auditlogboeken met succes/mislukking en sleuteltoeschrijving | -| 🔐**MCP-scopehandhaving** | 10 gedetailleerde scope-machtigingen voor gecontroleerde toegang tot tools | -| 📡**A2A Taaklevenscyclusbeheer** | Lijst/filter taken, inspecteer gebeurtenissen/artefacten, annuleer lopende taken | -| 📋**Agentenkaart ontdekken** | `/.well-known/agent.json` voor automatische detectie van clients | -| 🧪**Protocol E2E-testharnas** | Echte MCP SDK + A2A-clientstromen in `test:protocols:e2e` | -| ⚙️**Operationele controles** | Wissel van combo, pas veerkrachtprofielen toe, reset onderbrekers vanaf één bedieningsoppervlak | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Kenmerk | Wat het doet | -| ------------------------------------------ | --------------------------------------------------------------------------------------- | ----------------------- | -| 🎯**Slimme terugval op 4 niveaus** | Autoroute: Abonnement → API-sleutel → Goedkoop → Gratis | -| 📊**Realtime quota bijhouden** | Live tokentelling + reset-aftelling per provider | -| 🔄**Formaatvertaling** | OpenAI ↔ Claude ↔ Gemini ↔ Reacties met schemaveilige conversies | -| 👥**Ondersteuning voor meerdere accounts** | Meerdere accounts per aanbieder met intelligente selectie | -| 🔄**Automatische tokenvernieuwing** | OAuth-tokens worden automatisch vernieuwd bij nieuwe poging | -| 🎨**Aangepaste combo's** | 9 balanceringsstrategieën + terugvalketencontrole | -| 🌐**Wildcard-router** | `provider/*` dynamische routering | -| 🧠**Begrotingscontroles bedenken** | Passthrough-, automatische, aangepaste en adaptieve redeneerlimieten | -| 🔀**Modelaliassen** | Ingebouwde + aangepaste modelaliasing en migratieveiligheid | -| ⚡**Achtergronddegradatie** | Achtergrondtaken met lage prioriteit routeren naar goedkopere modellen | -| 🧪**Taakbewuste slimme routering** | Model automatisch selecteren op inhoudstype (codering/visie/analyse/samenvatting) | -| 🔄**A2A-agentworkflows** | Deterministische FSM-orkestrator voor stateful meerstaps agentuitvoeringen | -| 🔀**Adaptieve routering** | Dynamische strategie-overschrijving op basis van tokenvolume en promptcomplexiteit | -| 🎲**Diversiteit van aanbieders** | Shannon-entropiescore balanceert automatische combo-verkeersdistributie | -| 💬**Systeempromptinjectie** | Wereldwijde gedragscontroles consequent toegepast | -| 📄**Reacties API-compatibiliteit** | Volledige `/v1/responses` ondersteuning voor Codex en geavanceerde agentische workflows | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Kenmerk | Wat het doet | -| ----------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**Beeld genereren** | `/v1/images/generations` met cloud- en lokale backends | -| 📐**Insluitingen** | `/v1/embeddings` voor zoek- en RAG-pijplijnen | -| 🎤**Audiotranscriptie** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), automatische taaldetectie, MP4/MP3/WAV-ondersteuning | -| 🔊**Tekst-naar-spraak** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) met correcte foutmeldingen | -| 🎬**Videogeneratie** | `/v1/videos/generations` (ComfyUI + SD WebUI-workflows) | -| 🎵**Muziekgeneratie** | `/v1/music/generations` (ComfyUI-workflows) | -| 🛡️**Moderaties** | `/v1/moderations` veiligheidscontroles | -| 🔀**Herschikking** | `/v1/rerank` voor relevantiescore | -| 🔍**Webzoeken**🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6.500+ gratis/maand, automatische failover, cache | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Kenmerk | Wat het doet | -| ------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**Stroomonderbrekers** | Trip/herstel per model met drempelcontroles | -| 🎯**Eindpuntbewuste modellen** | Aangepaste modellen declareren ondersteunde eindpunten + API-indeling | -| 🛡️**Anti-donderende kudde** | Mutex + semafoorbescherming bij nieuwe pogingen/beoordeelgebeurtenissen | -| 🧠**Semantische + handtekeningcache** | Kosten-/latentiereductie met twee cachelagen | -| ⚡**Idempotentie aanvragen** | Dubbelbeschermingsvenster | -| 🔒**TLS-vingerafdrukspoofing** | Browserachtige TLS-vingerafdruk —**vermindert botdetectie en accountmarkering** | -| 🔏**CLI-vingerafdrukmatching** | Komt overeen met de handtekeningen van native CLI-verzoeken —**vermindert het verbodsrisico terwijl het proxy-IP behouden blijft** | -| 🌐**IP-filtering** | Toelatingslijst/blokkeerlijstbeheer voor blootgestelde implementaties | -| 📊**Bewerkbare tarieflimieten** | Configureerbare limieten op globaal/providerniveau met persistentie | -| 📉**Sierlijke degradatie** | Fallbacks met meerlaagse mogelijkheden ter bescherming van kerngatewaybewerkingen | -| 📜**Config-audittraject** | Op verschillen gebaseerde tracking van wijzigingen voorkomt operationele drift met eenvoudige rollbacks | -| ⏳**Provider Health Sync** | Proactieve monitoring van de vervaldatum van tokens, waardoor waarschuwingen worden geactiveerd voordat autorisatie mislukt | -| 🚪**Verbannen accounts automatisch uitschakelen** | Operationele stroomonderbreker die permanent geblokkeerde token-accounts automatisch verzegelt | -| 🔑**API-sleutelbeheer + Scoping** | Veilige sleuteluitgifte/roulatie en controles op modellen/aanbieders | -| 👁️**Scoped API Key Reveal**🆕 | Meld u aan voor herstel van API-sleutels via `ALLOW_API_KEY_REVEAL` | -| 🛡️**Beschermde `/modellen`** | Optionele authenticatie en providerverberging voor modelcatalogus | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Kenmerk | Wat het doet | -| ------------------------------------------- | ----------------------------------------------------------------------- | ---------------------------- | -| 📝**Verzoek + proxyregistratie** | Volledige aanvraag/antwoord en proxyregistratie | -| 📉**Gestreamde gedetailleerde logboeken**🆕 | Reconstrueert SSE-payloadstromen netjes in de gebruikersinterface | -| 📋**Unified Logdashboard** | Verzoek-, proxy-, audit- en consoleweergaven op één pagina | -| 🔍**Telemetrie aanvragen** | p50/p95/p99-latentie en tracering van aanvragen | -| 🏥**Gezondheidsdashboard** | Uptime, status van stroomonderbrekers, uitsluitingen, cachestatistieken | -| 💰**Kosten bijhouden** | Budgetcontroles en prijszichtbaarheid per model | -| 📈**Analytische visualisaties** | Gebruiksinzichten in modellen/aanbieders en trendweergaven | -| 🧪**Evaluatiekader** | Golden set-testen met configureerbare wedstrijdstrategieën | -| 📡**Live diagnostiek**🆕 | Semantische cache-bypass voor nauwkeurige combo-livetests | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Kenmerk | Wat het doet | -| -------------------------------- | ------------------------------------------------------------------------ | --------------------- | -| 🌐**Overal implementeren** | Localhost, VPS, Docker, Cloudomgevingen | -| 🚇**Cloudflare-tunnel**🆕 | Snelle Tunnel-integratie met één klik vanaf het dashboard | -| 🔑**API-sleutelmodelfiltering** | Native /v1/models-antwoord gefilterd via toegewezen Bearer-contextrollen | -| ⚡**Smart Cache Bypass** | Configureerbare TTL-heuristieken en geforceerde ophaalbedieningen | -| 🔄**Back-up/herstellen** | Export/import en noodherstelstromen | -| 🧙**Onboarding-wizard** | Begeleide installatie bij eerste gebruik | -| 🔧**CLI Tools-dashboard** | Installatie met één klik voor populaire codeertools | -| 🎮**Modelspeeltuin** | Test elke provider/model/eindpunt vanaf het dashboard | -| 🔏**CLI-vingerafdrukschakelaar** | Matching van vingerafdrukken per provider in Instellingen > Beveiliging | -| 🌐**i18n (30 talen)** | Volledig dashboard + documenttaalondersteuning met RTL-dekking | -| 🧹**Alle modellen wissen** | Modellijst wissen met één klik in providergegevens | -| 👁️**Zijbalkbediening**🆕 | Componenten en integraties verbergen in Weergave-instellingen | -| 📋**Uitgiftesjablonen** | Gestandaardiseerde GitHub-sjablonen voor bugs en functies | -| 📂**Aangepaste gegevensmap** | `DATA_DIR` overschrijven voor opslaglocatie | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1292,103 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -Wanneer quota, tarief of gezondheid falen, gaat OmniRoute automatisch naar de volgende kandidaat zonder handmatig te schakelen.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A zijn vindbaar in de gebruikersinterface en documenten (niet verborgen) -- Protocolstatus-API's geven live operationele gegevens weer (`/api/mcp/*`, `/api/a2a/*`) -- Dashboards bevatten acties voor operaties van dag 2 (combinatieschakelaars, reset van onderbrekers, annulering van taken)#### Translator + validation workflow +#### Protocol management that is visible and operable -Het Vertalergebied omvat: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Speeltuin**: transformatiecontroles aanvragen -**Chat Tester**: volledige aanvraag/antwoordrondreis -**Testbank**: meerdere cases in één run -**Live Monitor**: realtime verkeersinformatie +#### Translator + validation workflow -Plus protocolvalidatie met echte clients via `npm run test:protocols:e2e`. +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Toolreferentie, IDE-configuraties en clientvoorbeelden +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A Server README](src/lib/a2a/README.md)**— Vaardigheden, JSON-RPC-methoden, streaming en taaklevenscyclus## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute bevat een ingebouwd evaluatieframework om de LLM-responskwaliteit te testen aan de hand van een gouden set. U kunt deze openen via**Analytics → Evaluaties**in het dashboard.### Built-in Golden Set +## 🧪 Evaluations (Evals) -De vooraf geladen "OmniRoute Golden Set" bevat testcases voor: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Groeten, wiskunde, aardrijkskunde, codegeneratie -- Naleving van JSON-formaten, vertaling, genereren van prijsverlagingen -- Veiligheidsweigering (schadelijke inhoud), tellen, booleaanse logica### Evaluation Strategies +### Built-in Golden Set -| Strategie | Beschrijving | Voorbeeld | -| --------- | --------------------------------------------------------------------- | ---------------------------------- | --- | -| `exact` | De uitvoer moet exact overeenkomen met | `"4"` | -| `bevat` | De uitvoer moet een subtekenreeks bevatten (niet hoofdlettergevoelig) | `"Parijs"` | -| `regex` | Uitvoer moet overeenkomen met regex-patroon | `"1.*2.*3"` | -| `op maat` | Aangepaste JS-functie retourneert waar/onwaar | `(uitvoer) => uitvoer.lengte > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A)
-🧩 MCP-installatie (Model Context Protocol) +🧩 MCP Setup (Model Context Protocol) -Start MCP-transport in stdio-modus:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Aanbevolen validatiestroom: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Verbind uw MCP-client via stdio. -2. Voer `omniroute_get_health` uit. -3. Voer 'omniroute_list_combos' uit. -4. Open `/dashboard/mcp` om hartslag, activiteit en audit te bevestigen. +Useful APIs for automation: -Handige API's voor automatisering: - -- `KRIJG /api/mcp/status` -- `Haal /api/mcp/tools` +- `GET /api/mcp/status` +- `GET /api/mcp/tools` - `GET /api/mcp/audit` -- `Haal /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats` + +
-🤝 A2A-installatie (Agent2Agent) +🤝 A2A Setup (Agent2Agent) -Ontdek de makelaar:```bash +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Stuur een taak:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` +Manage lifecycle: -Levenscyclus beheren: - -- `KRIJG /api/a2a/status` +- `GET /api/a2a/status` - `GET /api/a2a/tasks` -- `Haal /api/a2a/tasks/:id` +- `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -Operationele gebruikersinterface: +Operational UI: -- `/dashboard/a2a` voor observatie van taken/status/stream en rookacties
+- `/dashboard/a2a` for task/state/stream observability and smoke actions + +
-🧪 End-to-end protocolvalidatie +🧪 End-to-end protocol validation -Valideer beide protocollen met echte clients:```bash +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Dit verifieert: +This verifies: -- MCP SDK-client verbinden/lijst/oproepen -- A2A-detectie/verzenden/streamen/ophalen/annuleren -- Controleer gegevens in MCP-audit en A2A-taakbeheer-API's
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs + +
-💳 Abonnementaanbieders### Claude Code (Pro/Max) +💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1401,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Pro-tip:**Gebruik Opus voor complexe taken, Sonnet voor snelheid. OmniRoute houdt quota bij per model!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1415,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Elk Codex-account heeft nu beleidsschakelaars in `Dashboard -> Providers`: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (AAN/UIT): handhaaf het drempelbeleid van 5 uur. -- `Wekelijks` (AAN/UIT): het drempelbeleid voor wekelijkse vensters afdwingen. -- Drempelgedrag: wanneer een ingeschakeld venster >=90% gebruik bereikt, wordt dat account overgeslagen. -- Rotatiegedrag: OmniRoute routeert automatisch naar het volgende in aanmerking komende Codex-account. -- Resetgedrag: wanneer de `resetAt`-tijd van de provider verstrijkt, komt het account automatisch weer in aanmerking. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Scenario's: +Scenarios: -- `5u AAN` + `Wekelijks AAN`: account wordt overgeslagen wanneer een van beide vensters de drempel bereikt. -- `5u UIT` + `Wekelijks AAN`: alleen wekelijks gebruik kan het account blokkeren. -- `5u AAN` + `Wekelijks UIT`: alleen 5 uur gebruik kan het account blokkeren. -- `resetAt` doorgegeven: account gaat automatisch opnieuw in rotatie (geen handmatige herinschakeling).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1440,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Beste waarde:**Enorm gratis niveau! Gebruik dit vóór betaalde niveaus.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1456,70 +1663,90 @@ Models:
-🔑 API-sleutelproviders### NVIDIA NIM (FREE developer access — 70+ models) +🔑 API Key Providers -1. Aanmelden: [build.nvidia.com](https://build.nvidia.com) -2. Ontvang een gratis API-sleutel (inclusief 1000 inferentiecredits) -3. Dashboard → Provider toevoegen → NVIDIA NIM: - - API-sleutel: `nvapi-uw-sleutel` +### NVIDIA NIM (FREE developer access — 70+ models) -**Modellen:**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` en 50+ meer +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Pro-tip:**OpenAI-compatibele API — werkt naadloos samen met de formaatvertaling van OmniRoute!### DeepSeek +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -1. Aanmelden: [platform.deepseek.com](https://platform.deepseek.com) -2. Haal de API-sleutel op -3. Dashboard → Provider toevoegen → DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -**Modellen:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +### DeepSeek -1. Aanmelden: [console.groq.com](https://console.groq.com) -2. Ontvang een API-sleutel (inclusief gratis laag) -3. Dashboard → Provider toevoegen → Groq +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -**Modellen:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Pro-tip:**Ultrasnelle gevolgtrekking — het beste voor realtime coderen!### OpenRouter (100+ Models) +### Groq (Free Tier Available!) -1. Aanmelden: [openrouter.ai](https://openrouter.ai) -2. Haal de API-sleutel op -3. Dashboard → Provider toevoegen → OpenRouter +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -**Modellen:**Krijg toegang tot meer dan 100 modellen van alle grote providers via één enkele API-sleutel. +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Dashboardgedrag:**OpenRouter-modellen worden beheerd vanuit**Beschikbare modellen**. Handmatig toevoegen, importeren en automatisch synchroniseren werken allemaal dezelfde lijst bij.
+**Pro Tip:** Ultra-fast inference — best for real-time coding! + +### OpenRouter (100+ Models) + +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter + +**Models:** Access 100+ models from all major providers through a single API key. + +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. + +
-💰 Goedkope providers (back-up)### GLM-4.7 (Daily reset, $0.6/1M) +💰 Cheap Providers (Backup) -1. Aanmelden: [Zhipu AI](https://open.bigmodel.cn/) -2. Haal de API-sleutel op uit het Coderingsplan -3. Dashboard → API-sleutel toevoegen: - - Aanbieder: `glm` - - API-sleutel: `uw-sleutel` +### GLM-4.7 (Daily reset, $0.6/1M) -**Gebruik:**`glm/glm-4.7` +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**Pro-tip:**Coderingsplan biedt 3× quotum tegen 1/7 kosten! Dagelijks resetten om 10:00 uur.### MiniMax M2.1 (5h reset, $0.20/1M) +**Use:** `glm/glm-4.7` -1. Aanmelden: [MiniMax](https://www.minimax.io/) -2. Haal de API-sleutel op -3. Dashboard → API-sleutel toevoegen +**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. -**Gebruik:**`minimax/MiniMax-M2.1` +### MiniMax M2.1 (5h reset, $0.20/1M) -**Pro-tip:**Goedkoopste optie voor lange context (1 miljoen tokens)!### Kimi K2 ($9/month flat) +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key -1. Abonneer je op: [Moonshot AI](https://platform.moonshot.ai/) -2. Haal de API-sleutel op -3. Dashboard → API-sleutel toevoegen +**Use:** `minimax/MiniMax-M2.1` -**Gebruik:**`kimi/kimi-latest` +**Pro Tip:** Cheapest option for long context (1M tokens)! -**Pro-tip:**Vaste $ 9/maand voor 10 miljoen tokens = $ 0,90/1 miljoen effectieve kosten!
+### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + +
-🆓 GRATIS providers (noodback-up)### Qoder (5 FREE models via OAuth) +🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1561,7 +1788,9 @@ Models:
-🎨 Combo's maken### Example 1: Maximize Subscription → Cheap Backup +🎨 Create Combos + +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1590,7 +1819,9 @@ Cost: $0 forever!
-🔧 CLI-integratie### Cursor IDE +🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1601,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Gebruik de pagina**CLI Tools**in het dashboard voor configuratie met één klik, of bewerk `~/.claude/settings.json` handmatig.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1612,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Optie 1 — Dashboard (aanbevolen):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Optie 2 — Handmatig:**Bewerk `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1629,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Opmerking:**OpenClaw werkt alleen met lokale OmniRoute. Gebruik `127.0.0.1` in plaats van `localhost` om IPv6-resolutieproblemen te voorkomen.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1643,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Stap 1:**Voeg OmniRoute toe als aangepaste provider:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Stap 2:**Maak/bewerk `opencode.json` in de hoofdmap van uw project:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1669,117 +1909,130 @@ opencode } } } -```` +``` -**Stap 3:**Selecteer het model in OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Tip:**Voeg elk model dat beschikbaar is in uw OmniRoute `/v1/models`-eindpunt toe aan de sectie `modellen`. Gebruik het formaat `provider/model-id` van uw OmniRoute-dashboard.
+ --- ## Probleemoplossing
-Klik om de probleemoplossingsgids uit te vouwen +Click to expand troubleshooting guide -**"Taalmodel heeft geen berichten verstrekt"** +**"Language model did not provide messages"** -- Providerquotum opgebruikt → Controleer dashboardquotumtracker -- Oplossing: gebruik combo-fallback of schakel over naar een goedkoper niveau +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**Snelheidslimiet** +**Rate limiting** -- Abonnementquotum op → Terugval op GLM/MiniMax -- Combo toevoegen: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**OAuth-token verlopen** +**OAuth token expired** -- Automatisch vernieuwd door OmniRoute -- Als de problemen aanhouden: Dashboard → Provider → Opnieuw verbinding maken +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Hoge kosten** +**High costs** -- Controleer gebruiksstatistieken in Dashboard → Kosten -- Schakel het primaire model over naar GLM/MiniMax -- Gebruik de gratis laag (Gemini CLI, Qoder) voor niet-kritieke taken +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Dashboard-/API-poorten zijn verkeerd** +**Dashboard/API ports are wrong** -- `POORT` is de canonieke basispoort (en standaard de API-poort) -- `API_PORT` overschrijft alleen de OpenAI-compatibele API-listener -- `DASHBOARD_PORT` overschrijft alleen dashboard/Next.js-listener -- Stel `NEXT_PUBLIC_BASE_URL` in op uw dashboard/openbare URL (voor OAuth-callbacks) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Cloudsynchronisatiefouten** +**Cloud sync errors** -- Controleer of `BASE_URL` verwijst naar uw actieve exemplaar -- Controleer of `CLOUD_URL` verwijst naar uw verwachte cloudeindpunt -- Houd de `NEXT_PUBLIC_*`-waarden uitgelijnd met de waarden op de server +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Eerste login werkt niet** +**First login not working** -- Controleer `INITIAL_PASSWORD` in `.env` -- Indien niet ingesteld, is het reservewachtwoord '123456' +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Geen verzoeklogboeken** +**No request logs** -- Verzoekartefacten worden naar `DATA_DIR/call_logs/` geschreven als één JSON-bestand per verzoek -- Schakel het vastleggen van pijplijnen in vanuit Dashboard → Logboeken → Logboeken aanvragen als u gedetailleerde payloads per fase nodig heeft -- Stel `APP_LOG_TO_FILE=true` in als u ook logboeken van de applicatieconsole wilt in `logs/application/app.log` -- Pas indien nodig `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` en `CALL_LOG_MAX_ENTRIES` aan +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Verbindingstest toont "Ongeldig" voor OpenAI-compatibele providers** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Veel providers stellen geen `/modellen`-eindpunt beschikbaar -- OmniRoute v1.0.6+ omvat fallback-validatie via chat-voltooiingen -- Zorg ervoor dat de basis-URL het achtervoegsel `/v1` bevat### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix - +### 🔐 OAuth on a Remote Server + + ->**⚠️ Belangrijk voor gebruikers die OmniRoute gebruiken op een VPS, Docker of een externe server**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -De providers**Antigravity**en**Gemini CLI**gebruiken**Google OAuth 2.0**. Google vereist dat de `redirect_uri` in de OAuth-stroom exact overeenkomt met een van de vooraf geregistreerde URI's in de Google Cloud Console van de app. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -De OAuth-referenties die in OmniRoute zijn gebundeld, zijn**alleen voor `localhost`**geregistreerd. Wanneer u OmniRoute opent op een externe server (bijvoorbeeld `https://omniroute.myserver.com`), weigert Google de authenticatie met:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -U moet een**OAuth 2.0 Client ID**maken in Google Cloud Console met de URI van uw server.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. + +#### Step-by-step **1. Open Google Cloud Console** -Ga naar: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -**2. Maak een nieuwe OAuth 2.0-client-ID** +**2. Create a new OAuth 2.0 Client ID** -- Klik op**"+ Credentials aanmaken"**→**"OAuth-client-ID"** -- Applicatietype:**"Webapplicatie"** -- Naam: wat je maar wilt (bijvoorbeeld `OmniRoute Remote`) +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -**3. Geautoriseerde omleidings-URI's toevoegen** +**3. Add Authorized Redirect URIs** -Voeg in het veld**"Geautoriseerde omleidings-URI's"**het volgende toe:``` +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Vervang `uw-server.com` door het domein of IP-adres van uw server (voeg indien nodig de poort toe, bijvoorbeeld `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. Bewaar en kopieer de inloggegevens** +After creating, Google will show the **Client ID** and **Client Secret**. -Na het maken toont Google de**Client-ID**en**Clientgeheim**. +**5. Set environment variables** -**5. Omgevingsvariabelen instellen** +In your `.env` (or Docker environment variables): -In uw `.env` (of Docker-omgevingsvariabelen):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1788,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Start OmniRoute opnieuw**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Probeer opnieuw verbinding te maken** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Dashboard → Providers → Antigravity (of Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google zal nu correct doorverwijzen naar `https://uw-server.com/callback`.--- +--- #### Temporary workaround (without custom credentials) -Als u nu niet uw eigen inloggegevens wilt instellen, kunt u nog steeds de**handmatige URL-stroom**gebruiken: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute opent de Google-autorisatie-URL -2. Na autorisatie probeert Google om te leiden naar `localhost` (wat mislukt op de externe server) -3.**Kopieer de volledige URL**uit de adresbalk van uw browser (zelfs als de pagina niet wordt geladen) -4. Plak die URL in het veld dat wordt weergegeven in het OmniRoute-verbindingsmodel -5. Klik op**"Verbinden"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Dit werkt omdat de autorisatiecode in de URL geldig is, ongeacht of de omleidingspagina is geladen.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. + +---
-🇧🇷 Versão em Português#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +🇧🇷 Versão em Português -Deze bewijzen**Antigravity**en**Gemini CLI**gebruiken**Google OAuth 2.0**voor authenticatie. Of Google zegt dat een `redirect_uri` geen OAuth-zoekopdracht**exatamente**een van de URI's vóór de kadaster heeft uitgevoerd zonder dat de Google Cloud Console een toepassing heeft. +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? -Omdat OAuth geen OmniRoute-registratie heeft, is dit**apenas para `localhost`**. Wanneer u OmniRoute op een externe server opent (bijvoorbeeld: `https://omniroute.meuservidor.com`), of Google een autenticação com:``` +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -U kunt precies zien hoe**OAuth 2.0 Client ID**geen Google Cloud Console heeft met een URI van zijn server.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Toegang tot Google Cloud Console** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -**2. Nieuwe OAuth 2.0 client-ID** +**2. Crie um novo OAuth 2.0 Client ID** -- Klik op**"+ Credentials aanmaken"**→**"OAuth-client-ID"** -- Applicatietip:**"Webapplicatie"** -- Nome: escolha qualquer nome (bijvoorbeeld: `OmniRoute Remote`) +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Adicione als geautoriseerde omleidings-URI's** +**3. Adicione as Authorized Redirect URIs** -Geen campagne**"Geautoriseerde omleidings-URI's"**, aanbevolen:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Vervang `seu-servidor.com` door de domicilie of het IP-adres van uw server (inclusief een noodzakelijke porta, bijvoorbeeld: `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. Bewaar en kopieer als credenciais** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -U kunt ook op Google klikken op**Client-ID**en**Clientgeheim**. +**5. Configure as variáveis de ambiente** -**5. Configureer als variáveis de ambiente** +No seu `.env` (ou nas variáveis de ambiente do Docker): -Geen `.env` (of de verschillende omgevingen van Docker):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1867,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reinicie van OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute +``` -```` - -**7. Nieuwe verbinding** +**7. Tente conectar novamente** Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Nadat Google de juiste verwijzing naar `https://seu-servidor.com/callback` heeft gemaakt en een automatische functie heeft uitgevoerd.--- +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. + +--- #### Workaround temporário (sem configurar credenciais próprias) -Als u geen geloofwaardige geloofwaardigheid meer heeft, is het mogelijk om de stroom**handleiding van de URL**te gebruiken: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. OmniRoute gebruikt een autorisatie-URL van Google -2. Nadat u de autorisatie heeft gegeven, zal Google de verwijzing naar 'localhost' doorsturen (die geen externe service biedt) -3.**Kopieer een volledige URL**door de browser van uw browser (het bericht dat de pagina niet verder gaat) -4. Cole essa URL is niet beschikbaar op de verbindingswijze van OmniRoute -5. Klik op**"Verbinden"** +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) +4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute +5. Clique em **"Connect"** -> Deze tijdelijke oplossing werkt door de autorisatiecode van de URL en is onafhankelijk van het omleiden naar uw autorisatie of niet.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1906,63 +2172,72 @@ Als u geen geloofwaardige geloofwaardigheid meer heeft, is het mogelijk om de st ## 🛠️ Tech Stack
-Klik om de details van de tech-stack uit te vouwen +Click to expand tech stack details --**Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ wordt**niet ondersteund**— `better-sqlite3` native binaire bestanden zijn incompatibel) --**Taal**: TypeScript 5.9 —**100% TypeScript**voor `src/` en `open-sse/` (nul `any` in kernmodules sinds v2.0) --**Framework**: Next.js 16 + React 19 + Tailwind CSS 4 --**Database**: LowDB (JSON) + SQLite (domeinstatus + proxylogboeken + MCP-audit + routeringsbeslissingen) --**Schema's**: Zod (MCP-tool I/O-validatie, API-contracten) --**Protocollen**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Streaming**: door de server verzonden gebeurtenissen (SSE) --**Auth**: OAuth 2.0 (PKCE) + JWT + API-sleutels + MCP-scoped autorisatie --**Testen**: Node.js testrunner + Vitest (900+ tests inclusief unit, integratie, E2E) --**CI/CD**: GitHub-acties (automatische npm-publicatie + Docker Hub bij release) --**Website**: [omniroute.online](https://omniroute.online) --**Pakket**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Veerkracht**: stroomonderbreker, exponentiële uitstel, anti-donderkudde, TLS-spoofing, automatische combo-zelfherstel
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Documentatie -| Document | Beschrijving | -| ----------------------------------------- | ---------------------------------------------- | -| [Gebruikershandleiding](docs/USER_GUIDE.md) | Providers, combo's, CLI-integratie, implementatie | -| [API-referentie](docs/API_REFERENCE.md) | Alle eindpunten met voorbeelden | -| [MCP-server](open-sse/mcp-server/README.md) | 16 MCP-tools, IDE-configuraties, Python/TS/Go-clients | -| [A2A-server](src/lib/a2a/README.md) | JSON-RPC 2.0-protocol, vaardigheden, streaming, taakbeheer | -| [Auto-Combo-engine](docs/auto-combo.md) | Scoren met 6 factoren, moduspakketten, zelfgenezing | -| [Problemen oplossen](docs/TROUBLESHOOTING.md) | Veelvoorkomende problemen en oplossingen | -| [Architectuur](docs/ARCHITECTURE.md) | Systeemarchitectuur en internals | -| [Bijdragen](CONTRIBUTING.md) | Ontwikkelingsopstelling en richtlijnen | -| [OpenAPI-specificatie](docs/openapi.yaml) | OpenAPI 3.0-specificatie | -| [Beveiligingsbeleid](SECURITY.md) | Kwetsbaarheidsrapportage en beveiligingspraktijken | -| [VM-implementatie](docs/VM_DEPLOYMENT_GUIDE.md) | Volledige gids: VM + nginx + Cloudflare-installatie | -| [Functiegalerij](docs/FEATURES.md) | Visuele dashboardrondleiding met screenshots | -| [Releasechecklist](docs/RELEASE_CHECKLIST.md) | Validatiestappen vóór de release |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute heeft**210+ functies gepland**over meerdere ontwikkelingsfasen. Dit zijn de belangrijkste gebieden: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Categorie | Geplande functies | Hoogtepunten | -| --------------------------- | ---------------- | ----------------------------------------------------------------------------- | -| 🧠**Routing en intelligentie**| 25+ | Routering met de laagste latentie, op tags gebaseerde routering, quota-preflight, P2C-accountselectie | -| 🔒**Beveiliging en naleving**| 20+ | SSRF-verharding, cloaking van inloggegevens, snelheidslimiet per eindpunt, scoping van beheersleutels | -| 📊**Waarneembaarheid**| 15+ | OpenTelemetry-integratie, realtime quotabewaking, kostenregistratie per model | -| 🔄**Provider-integraties**| 20+ | Dynamisch modelregister, cooldowns van providers, Codex met meerdere accounts, parseren van Copilot-quota | -| ⚡**Prestaties**| 15+ | Dubbele cachelaag, promptcache, responscache, streaming keepalive, batch-API | -| 🌐**Ecosysteem**| 10+ | WebSocket API, configuratie hot-reload, gedistribueerde configuratieopslag, commerciële modus |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**OpenCode-integratie**— Native providerondersteuning voor de OpenCode AI-coderings-IDE -- 🔗**TRAE-integratie**— Volledige ondersteuning voor het TRAE AI-ontwikkelingsframework -- 📦**Batch API**— Asynchrone batchverwerking voor bulkaanvragen -- 🎯**Op tags gebaseerde routering**— Routeer verzoeken op basis van aangepaste tags en metagegevens -- 💰**Laagste kostenstrategie**— Selecteer automatisch de goedkoopste beschikbare provider +### 🔜 Coming Soon -> 📝 Volledige functiespecificaties beschikbaar in [`docs/new-features/`](docs/new-features/) (217 gedetailleerde specificaties)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1970,18 +2245,20 @@ OmniRoute heeft**210+ functies gepland**over meerdere ontwikkelingsfasen. Dit zi ### How to Contribute -1. Fork de repository -2. Maak je feature branch (`git checkout -b feature/amazing-feature`) -3. Voer je wijzigingen door (`git commit -m 'Geweldige functie toevoegen'`) -4. Push naar de branch (`git push origin feature/amazing-feature`) -5. Open een Pull Request +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Zie [CONTRIBUTING.md](CONTRIBUTING.md) voor gedetailleerde richtlijnen.### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1993,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Speciale dank aan**[9router](https://github.com/decolua/9router)**door**[decolua](https://github.com/decolua)**— het originele project dat deze vork inspireerde. OmniRoute bouwt voort op die ongelooflijke basis met extra functies, multimodale API's en een volledige TypeScript-herschrijving. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Speciale dank aan**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— de originele Go-implementatie die deze JavaScript-poort inspireerde.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Licentie -MIT-licentie - zie [LICENTIE](LICENTIE) voor details.--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/nl/docs/ARCHITECTURE.md b/docs/i18n/nl/docs/ARCHITECTURE.md index 07ae2c4ea1..14058f9e46 100644 --- a/docs/i18n/nl/docs/ARCHITECTURE.md +++ b/docs/i18n/nl/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Laatst bijgewerkt: 28-03-2026_## Executive Summary -OmniRoute is een lokale AI-routeringsgateway en dashboard gebouwd op Next.js. -Het biedt één OpenAI-compatibel eindpunt (`/v1/*`) en routeert verkeer over meerdere upstream-providers met vertaling, fallback, tokenvernieuwing en gebruiksregistratie. -Kernmogelijkheden: +_Last updated: 2026-03-28_ -- OpenAI-compatibel API-oppervlak voor CLI/tools (28 providers) -- Verzoek/antwoord-vertaling in verschillende providerformaten -- Modelcombo fallback (reeks met meerdere modellen) -- Terugval op accountniveau (meerdere accounts per provider) -- OAuth + API-sleutelproviderverbindingsbeheer -- Generatie van inbedding via `/v1/embeddings` (6 providers, 9 modellen) -- Beeldgeneratie via `/v1/images/generations` (4 providers, 9 modellen) -- Think-tag-parsing (`...`) voor redeneermodellen -- Reactieopschoning voor strikte OpenAI SDK-compatibiliteit -- Rolnormalisatie (ontwikkelaar → systeem, systeem → gebruiker) voor compatibiliteit tussen providers -- Gestructureerde uitvoerconversie (json_schema → Gemini responseSchema) -- Lokale persistentie voor providers, sleutels, aliassen, combo's, instellingen, prijzen -- Gebruik/kosten bijhouden en verzoekregistratie -- Optionele cloudsynchronisatie voor synchronisatie van meerdere apparaten/statussen -- IP-toelatingslijst/blokkeerlijst voor API-toegangscontrole -- Meedenken over budgetbeheer (passthrough/auto/custom/adaptive) -- Globale systeempromptinjectie -- Sessie volgen en vingerafdrukken maken -- Verbeterde tarieflimieten per account met providerspecifieke profielen -- Stroomonderbrekerpatroon voor veerkracht van de provider -- Bescherming tegen donderende kuddes met mutex-vergrendeling -- Op handtekeningen gebaseerde cache voor deduplicatie van verzoeken -- Domeinlaag: modelbeschikbaarheid, kostenregels, fallback-beleid, lock-outbeleid -- Persistentie van domeinstatus (SQLite-schrijfcache voor fallbacks, budgetten, uitsluitingen, stroomonderbrekers) -- Beleidsengine voor gecentraliseerde verzoekevaluatie (lockout → budget → fallback) -- Telemetrie aanvragen met p50/p95/p99-latency-aggregatie -- Correlatie-ID (X-Request-Id) voor end-to-end tracering -- Compliance-auditregistratie met opt-out per API-sleutel -- Evaluatiekader voor LLM-kwaliteitsborging -- Veerkracht UI-dashboard met realtime stroomonderbrekerstatus -- Modulaire OAuth-providers (12 individuele modules onder `src/lib/oauth/providers/`) +## Executive Summary -Primair runtimemodel: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- Next.js app-routes onder `src/app/api/*` implementeren zowel dashboard-API's als compatibiliteits-API's -- Een gedeelde SSE/routing-kern in `src/sse/*` + `open-sse/*` zorgt voor de uitvoering, vertaling, streaming, fallback en gebruik van de provider## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Runtime van lokale gateway -- Dashboardbeheer-API's -- Providerverificatie en tokenvernieuwing -- Vraag vertaling en SSE-streaming aan -- Lokale status + gebruikspersistentie -- Optionele cloudsynchronisatie-orkestratie### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Implementatie van cloudservices achter `NEXT_PUBLIC_CLOUD_URL` -- Provider SLA/controlevlak buiten het lokale proces -- Externe CLI-binaire bestanden zelf (Claude CLI, Codex CLI, enz.)## Dashboard Surface (Current) +### Out of Scope -Hoofdpagina's onder `src/app/(dashboard)/dashboard/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — snelle start + provideroverzicht -- `/dashboard/endpoint` — eindpuntproxy + MCP + A2A + API-eindpunttabbladen -- `/dashboard/providers` — providerverbindingen en inloggegevens -- `/dashboard/combos` — combostrategieën, sjablonen, modelrouteringsregels -- `/dashboard/kosten` — aggregatie van kosten en zichtbaarheid van prijzen -- `/dashboard/analytics` — gebruiksanalyses en evaluaties -- `/dashboard/limits` — controles op quota/tarieven -- `/dashboard/cli-tools` — CLI-onboarding, runtime-detectie, genereren van configuraties -- `/dashboard/agents` — gedetecteerde ACP-agenten + aangepaste agentregistratie -- `/dashboard/media` — speelplaats voor afbeeldingen/video/muziek -- `/dashboard/search-tools` — testen en geschiedenis van zoekmachines -- `/dashboard/health` — uptime, stroomonderbrekers, snelheidslimieten -- `/dashboard/logs` — verzoek/proxy/audit/console-logboeken -- `/dashboard/settings` — tabbladen met systeeminstellingen (algemeen, routing, combo-standaardinstellingen, enz.) -- `/dashboard/api-manager` — API-sleutellevenscyclus en modelrechten## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Hoofdmappen: +Main directories: -- `src/app/api/v1/*` en `src/app/api/v1beta/*` voor compatibiliteits-API's -- `src/app/api/*` voor beheer-/configuratie-API's -- Volgende herschrijving in `next.config.mjs` kaart `/v1/*` naar `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Belangrijke compatibiliteitsroutes: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` — bevat aangepaste modellen met `custom: true` -- `src/app/api/v1/embeddings/route.ts` — genereren van inbedden (6 providers) -- `src/app/api/v1/images/generations/route.ts` — generatie van afbeeldingen (4+ providers incl. Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — speciale chat per provider -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — speciale insluitingen per provider -- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — speciale afbeeldingen per provider +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` -- `src/app/api/v1beta/models/[...pad]/route.ts` +- `src/app/api/v1beta/models/[...path]/route.ts` -Beheerdomeinen: +Management domains: -- Authenticatie/instellingen: `src/app/api/auth/*`, `src/app/api/settings/*` -- Providers/verbindingen: `src/app/api/providers*` -- Providerknooppunten: `src/app/api/provider-nodes*` -- Aangepaste modellen: `src/app/api/provider-models` (GET/POST/DELETE) -- Modelcatalogus: `src/app/api/models/route.ts` (GET) -- Proxyconfiguratie: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- Sleutels/aliassen/combo's/prijzen: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- Gebruik: `src/app/api/usage/*` -- Synchronisatie/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` -- Hulpprogramma's voor CLI-tools: `src/app/api/cli-tools/*` -- IP-filter: `src/app/api/settings/ip-filter` (GET/PUT) -- Denkbudget: `src/app/api/settings/thinking-budget` (GET/PUT) -- Systeemprompt: `src/app/api/settings/system-prompt` (GET/PUT) -- Sessies: `src/app/api/sessions` (GET) -- Tarieflimieten: `src/app/api/rate-limits` (GET) -- Veerkracht: `src/app/api/resilience` (GET/PATCH) — providerprofielen, stroomonderbreker, snelheidslimietstatus -- Veerkracht reset: `src/app/api/resilience/reset` (POST) - reset onderbrekers + cooldowns -- Cachestatistieken: `src/app/api/cache/stats` (GET/DELETE) -- Beschikbaarheid van modellen: `src/app/api/models/availability` (GET/POST) -- Telemetrie: `src/app/api/telemetry/summary` (GET) +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) - Budget: `src/app/api/usage/budget` (GET/POST) -- Terugvalketens: `src/app/api/fallback/chains` (GET/POST/DELETE) -- Nalevingsaudit: `src/app/api/compliance/audit-log` (GET) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) - Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- Beleid: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Policies: `src/app/api/policies` (GET/POST) -Hoofdstroommodules: +## 2) SSE + Translation Core -- Invoer: `src/sse/handlers/chat.ts` -- Kernorkestratie: `open-sse/handlers/chatCore.ts` -- Uitvoeringsadapters van providers: `open-sse/executors/*` -- Formaatdetectie/providerconfiguratie: `open-sse/services/provider.ts` -- Model ontleden/oplossen: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Reservelogica voor accounts: `open-sse/services/accountFallback.ts` -- Vertaalregister: `open-sse/translator/index.ts` -- Streamtransformaties: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- Gebruiksextractie/normalisatie: `open-sse/utils/usageTracking.ts` -- Think tag-parser: `open-sse/utils/thinkTagParser.ts` -- Inbeddingshandler: `open-sse/handlers/embeddings.ts` -- Insluiten van providerregister: `open-sse/config/embeddingRegistry.ts` -- Handler voor het genereren van afbeeldingen: `open-sse/handlers/imageGeneration.ts` -- Register van de imageprovider: `open-sse/config/imageRegistry.ts` -- Opschoning van reacties: `open-sse/handlers/responseSanitizer.ts` -- Rolnormalisatie: `open-sse/services/roleNormalizer.ts` +Main flow modules: -Diensten (bedrijfslogica): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- Accountselectie/score: `open-sse/services/accountSelector.ts` -- Contextlevenscyclusbeheer: `open-sse/services/contextManager.ts` -- Handhaving van IP-filters: `open-sse/services/ipFilter.ts` -- Sessie volgen: `open-sse/services/sessionManager.ts` -- Ontdubbeling aanvragen: `open-sse/services/signatureCache.ts` -- Systeempromptinjectie: `open-sse/services/systemPrompt.ts` -- Denkend budgetbeheer: `open-sse/services/thinkingBudget.ts` -- Wildcard-modelroutering: `open-sse/services/wildcardRouter.ts` -- Beheer van tarieflimieten: `open-sse/services/rateLimitManager.ts` -- Stroomonderbreker: `open-sse/services/circuitBreaker.ts` +Services (business logic): -Domeinlaagmodules: +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- Beschikbaarheid van modellen: `src/lib/domain/modelAvailability.ts` -- Kostenregels/budgetten: `src/lib/domain/costRules.ts` -- Terugvalbeleid: `src/lib/domain/fallbackPolicy.ts` -- Combo-resolver: `src/lib/domain/comboResolver.ts` -- Uitsluitingsbeleid: `src/lib/domain/lockoutPolicy.ts` -- Beleidsengine: `src/domain/policyEngine.ts` — gecentraliseerde uitsluiting → budget → fallback-evaluatie -- Foutcodecatalogus: `src/lib/domain/errorCodes.ts` -- Verzoek-ID: `src/lib/domain/requestId.ts` -- Time-out voor ophalen: `src/lib/domain/fetchTimeout.ts` -- Telemetrie aanvragen: `src/lib/domain/requestTelemetry.ts` -- Naleving/audit: `src/lib/domain/compliance/index.ts` +Domain layer modules: + +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` - Eval runner: `src/lib/domain/evalRunner.ts` -- Persistentie van domeinstatus: `src/lib/db/domainState.ts` — SQLite CRUD voor fallback-ketens, budgetten, kostengeschiedenis, uitsluitingsstatus, stroomonderbrekers +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -OAuth-providermodules (12 individuele bestanden onder `src/lib/oauth/providers/`): +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -- Registerindex: `src/lib/oauth/providers/index.ts` -- Individuele providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` -- Thin wrapper: `src/lib/oauth/providers.ts` — exporteert opnieuw van individuele modules## 3) Persistence Layer +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -Primaire status DB (SQLite): +## 3) Persistence Layer -- Kerninfra: `src/lib/db/core.ts` (better-sqlite3, migraties, WAL) -- Façade opnieuw exporteren: `src/lib/localDb.ts` (dunne compatibiliteitslaag voor bellers) -- bestand: `${DATA_DIR}/storage.sqlite` (of `$XDG_CONFIG_HOME/omniroute/storage.sqlite` indien ingesteld, anders `~/.omniroute/storage.sqlite`) -- entiteiten (tabellen + KV-naamruimten): providerConnections, providerNodes, modelAliases, combo's, apiKeys, instellingen, prijzen,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +Primary state DB (SQLite): -Gebruikspersistentie: +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -- façade: `src/lib/usageDb.ts` (ontbonden modules in `src/lib/usage/*`) -- SQLite-tabellen in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` -- optionele bestandsartefacten blijven bestaan voor compatibiliteit/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- verouderde JSON-bestanden worden gemigreerd naar SQLite door opstartmigraties, indien aanwezig +Usage persistence: -Domeinstatus DB (SQLite): +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- `src/lib/db/domainState.ts` — CRUD-bewerkingen voor domeinstatus -- Tabellen (aangemaakt in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- Doorschrijfcachepatroon: kaarten in het geheugen zijn gezaghebbend tijdens runtime; mutaties worden synchroon naar SQLite geschreven; status wordt hersteld vanuit DB bij koude start## 4) Auth + Security Surfaces +Domain State DB (SQLite): -- Dashboardcookieverificatie: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- API-sleutel genereren/verificatie: `src/shared/utils/apiKey.ts` -- Providergeheimen bleven bestaan in `providerConnections`-vermeldingen -- Uitgaande proxy-ondersteuning via `open-sse/utils/proxyFetch.ts` (env vars) en `open-sse/utils/networkProxy.ts` (configureerbaar per provider of globaal)## 5) Cloud Sync +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync - Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Periodieke taak: `src/shared/services/cloudSyncScheduler.ts` -- Periodieke taak: `src/shared/services/modelSyncScheduler.ts` -- Beheerroute: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Fallback-beslissingen worden aangestuurd door `open-sse/services/accountFallback.ts` met behulp van statuscodes en heuristieken voor foutmeldingen. Combo-routering voegt een extra beveiliging toe: op de provider gerichte 400's, zoals upstream content-block- en rolvalidatiefouten, worden behandeld als model-lokale fouten, zodat latere combo-doelen nog steeds kunnen worden uitgevoerd.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -Vernieuwen tijdens live verkeer wordt uitgevoerd binnen `open-sse/handlers/chatCore.ts` via uitvoerder `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -Periodieke synchronisatie wordt geactiveerd door `CloudSyncScheduler` wanneer de cloud is ingeschakeld.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -Fysieke opslagbestanden: +Physical storage files: -- primaire runtime-DB: `${DATA_DIR}/storage.sqlite` -- logregels opvragen: `${DATA_DIR}/log.txt` (compat/debug-artefact) -- gestructureerde payload-archieven voor oproepen: `${DATA_DIR}/call_logs/` -- optionele vertaler/verzoek debug-sessies: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibiliteits-API's -- `src/app/api/v1/providers/[provider]/*`: speciale routes per provider (chat, insluitingen, afbeeldingen) -- `src/app/api/providers*`: provider CRUD, validatie, testen -- `src/app/api/provider-nodes*`: aangepast compatibel knooppuntbeheer -- `src/app/api/provider-models`: aangepast modelbeheer (CRUD) -- `src/app/api/models/route.ts`: modelcatalogus-API (aliassen + aangepaste modellen) -- `src/app/api/oauth/*`: OAuth/device-code stromen -- `src/app/api/keys*`: levenscyclus van lokale API-sleutel -- `src/app/api/models/alias`: aliasbeheer -- `src/app/api/combos*`: fallback-combobeheer -- `src/app/api/pricing`: prijsoverschrijvingen voor kostenberekening -- `src/app/api/settings/proxy`: proxyconfiguratie (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: uitgaande proxy-connectiviteitstest (POST) -- `src/app/api/usage/*`: API's voor gebruik en logboeken -- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloudsynchronisatie en cloudgerichte helpers -- `src/app/api/cli-tools/*`: lokale CLI-configuratieschrijvers/checkers -- `src/app/api/settings/ip-filter`: IP-toelatingslijst/blokkeerlijst (GET/PUT) -- `src/app/api/settings/thinking-budget`: thinking token budgetconfiguratie (GET/PUT) -- `src/app/api/settings/system-prompt`: globale systeemprompt (GET/PUT) -- `src/app/api/sessions`: actieve sessielijst (GET) -- `src/app/api/rate-limits`: tarieflimietstatus per account (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: verzoekparse, combo-afhandeling, accountselectielus -- `open-sse/handlers/chatCore.ts`: vertaling, verzending van de uitvoerder, afhandeling van nieuwe pogingen/vernieuwen, stream-instellingen -- `open-sse/executors/*`: providerspecifiek netwerk- en formaatgedrag### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: register en orkestratie van vertalers -- Vertalers aanvragen: `open-sse/translator/request/*` -- Antwoordvertalers: `open-sse/translator/response/*` -- Formaatconstanten: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: persistente configuratie/status en domeinpersistentie op SQLite -- `src/lib/localDb.ts`: compatibiliteit opnieuw exporteren voor DB-modules -- `src/lib/usageDb.ts`: gevel van gebruiksgeschiedenis/oproeplogboeken bovenop SQLite-tabellen## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Elke provider heeft een gespecialiseerde uitvoerder die `BaseExecutor` uitbreidt (in `open-sse/executors/base.ts`), die het bouwen van URL's, het bouwen van headers, opnieuw proberen met exponentiële backoff, hooks voor het vernieuwen van referenties en de `execute()` orkestratiemethode biedt. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| executeur | Aanbieder(s) | Speciale behandeling | -| --------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------- | -| `StandaardExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Verbijstering, Samen, Vuurwerk, Cerebras, Cohere, NVIDIA | Dynamische URL/header-configuratie per provider | -| `AntizwaartekrachtExecutor` | Google Antizwaartekracht | Aangepaste project-/sessie-ID's, opnieuw proberen na parseren | -| `CodexExecutor` | OpenAI-codex | Injecteert systeeminstructies, dwingt redeneerinspanning af | -| `CursorExecutor` | Cursor-IDE | ConnectRPC-protocol, Protobuf-codering, ondertekening aanvragen via checksum | -| `GithubExecutor` | GitHub-copiloot | Copilot-token vernieuwen, VSCode-nabootsende headers | -| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binair formaat → SSE-conversie | -| `GeminiCLIE-uitvoerder` | Tweeling CLI | Vernieuwingscyclus van Google OAuth-token | +### Persistence -Alle andere providers (inclusief aangepaste compatibele knooppunten) gebruiken de `DefaultExecutor`.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Aanbieder | Formaat | Autorisatie | Stroom | Niet-stream | Token vernieuwen | Gebruiks-API | -| ----------------- | ------------------ | ---------------------- | ---------------- | ----------- | ---------------- | -------------------------- | ------------------------------ | -| Claude | claude | API-sleutel / OAuth | ✅ | ✅ | ✅ | ⚠️Alleen beheerder | -| Tweeling | Tweeling | API-sleutel / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloudconsole | -| Tweeling CLI | tweeling-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloudconsole | -| Antizwaartekracht | anti-zwaartekracht | OAuth | ✅ | ✅ | ✅ | ✅ Volledige quota-API | -| Open AI | openai | API-sleutel | ✅ | ✅ | ❌ | ❌ | -| Codex | openai-reacties | OAuth | ✅ gedwongen | ❌ | ✅ | ✅ Tarieflimieten | -| GitHub-copiloot | openai | OAuth + Copilot-token | ✅ | ✅ | ✅ | ✅ Momentopnamen van quota | -| Cursor | cursor | Aangepaste controlesom | ✅ | ✅ | ❌ | ❌ | -| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Gebruikslimieten | -| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️Per aanvraag | -| Qoder | openai | OAuth (basis) | ✅ | ✅ | ✅ | ⚠️Per aanvraag | -| OpenRouter | openai | API-sleutel | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | claude | API-sleutel | ✅ | ✅ | ❌ | ❌ | -| DeepSeek | openai | API-sleutel | ✅ | ✅ | ❌ | ❌ | -| Groq | openai | API-sleutel | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | openai | API-sleutel | ✅ | ✅ | ❌ | ❌ | -| Mistral | openai | API-sleutel | ✅ | ✅ | ❌ | ❌ | -| Verbijstering | openai | API-sleutel | ✅ | ✅ | ❌ | ❌ | -| Samen AI | openai | API-sleutel | ✅ | ✅ | ❌ | ❌ | -| Vuurwerk AI | openai | API-sleutel | ✅ | ✅ | ❌ | ❌ | -| Hersenen | openai | API-sleutel | ✅ | ✅ | ❌ | ❌ | -| Cohier | openai | API-sleutel | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | openai | API-sleutel | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Gedetecteerde bronformaten zijn onder meer: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. + +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | + +All other providers (including custom compatible nodes) use the `DefaultExecutor`. + +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: - `openai` -- `openai-reacties` +- `openai-responses` - `claude` -- `Tweeling` +- `gemini` -Doelformaten zijn onder meer: +Target formats include: -- OpenAI-chat/reacties - -Claude -- Gemini/Gemini-CLI/Antigravity-envelop +- OpenAI chat/Responses +- Claude +- Gemini/Gemini-CLI/Antigravity envelope - Kiro - Cursor -Vertalingen gebruiken**OpenAI als hubformaat**— alle conversies gaan via OpenAI als tussenproduct:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Vertalingen worden dynamisch geselecteerd op basis van de vorm van de bronpayload en het doelformaat van de provider. +Additional processing layers in the translation pipeline: -Extra verwerkingslagen in de vertaalpijplijn: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Opschoning van reacties**— Verwijdert niet-standaardvelden uit reacties in OpenAI-formaat (zowel streaming als niet-streaming) om strikte SDK-naleving te garanderen --**Rolnormalisatie**— Converteert `ontwikkelaar` → `systeem` voor niet-OpenAI-doelen; voegt `systeem` → `gebruiker` samen voor modellen die de systeemrol afwijzen (GLM, ERNIE) --**Think-tagextractie**— Parseert `...`-blokken uit de inhoud in het `reasoning_content`-veld --**Gestructureerde uitvoer**— Converteert OpenAI `response_format.json_schema` naar Gemini's `responseMimeType` + `responseSchema`## Supported API Endpoints +## Supported API Endpoints -| Eindpunt | Formaat | Behandelaar | -| --------------------------------------------- | ------------------ | ----------------------------------------------------------- | -| `POST /v1/chat/voltooiingen` | OpenAI-chat | `src/sse/handlers/chat.ts` | -| `POST /v1/berichten` | Claude-berichten | Dezelfde handler (automatisch gedetecteerd) | -| `POST /v1/reacties` | OpenAI-reacties | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/embeddings` | OpenAI-insluitingen | `open-sse/handlers/embeddings.ts` | -| `GET /v1/embeddings` | Modellijst | API-route | -| `POST /v1/images/generations` | OpenAI-afbeeldingen | `open-sse/handlers/imageGeneration.ts` | -| `GET /v1/images/generations` | Modellijst | API-route | -| `POST /v1/providers/{provider}/chat/completions` | OpenAI-chat | Toegewijd per provider met modelvalidatie | -| `POST /v1/providers/{provider}/embeddings` | OpenAI-insluitingen | Toegewijd per provider met modelvalidatie | -| `POST /v1/providers/{provider}/images/generations` | OpenAI-afbeeldingen | Toegewijd per provider met modelvalidatie | -| `POST /v1/messages/count_tokens` | Claude-tokentelling | API-route | -| `GET /v1/models` | OpenAI-modellenlijst | API-route (chat + insluiten + afbeelding + aangepaste modellen) | -| `KRIJG /api/models/catalog` | Catalogus | Alle modellen gegroepeerd op aanbieder + type | -| `POST /v1beta/models/*:streamGenerateContent` | Gemini geboren | API-route | -| `GET/PUT/DELETE /api/settings/proxy` | Proxyconfiguratie | Netwerkproxyconfiguratie | -| `POST /api/settings/proxy/test` | Proxy-connectiviteit | Eindpunt proxystatus/connectiviteitstest | -| `GET/POST/DELETE /api/provider-modellen` | Providermodellen | Metagegevens van het providermodel ter ondersteuning van aangepaste en beheerde beschikbare modellen |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -De bypass-handler (`open-sse/utils/bypassHandler.ts`) onderschept bekende "throwaway"-verzoeken van Claude CLI - opwarmpingen, titelextracties en tokentellingen - en retourneert een**nepreactie**zonder upstream providertokens te verbruiken. Dit wordt alleen geactiveerd als `User-Agent` `claude-cli` bevat.## Request Logger Pipeline +## Bypass Handler -De verzoeklogger (`open-sse/utils/requestLogger.ts`) biedt een pijplijn voor het registreren van fouten in 7 fasen, standaard uitgeschakeld en ingeschakeld via `ENABLE_REQUEST_LOGS=true`:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Bestanden worden voor elke verzoeksessie naar `/logs//` geschreven.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- Afkoelperiode van provideraccount bij tijdelijke/snelheids-/authenticatiefouten -- accountterugval voordat het verzoek mislukt -- Terugval op combo-modellen wanneer het huidige model-/providerpad is uitgeput## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- vooraf controleren en vernieuwen met nieuwe poging voor vernieuwbare providers -- 401/403 opnieuw proberen na vernieuwingspoging in kernpad## 3) Stream Safety +## 2) Token Expiry -- verbindingsbewuste streamcontroller -- vertaalstroom met end-of-stream flush en `[DONE]` afhandeling -- Terugval in gebruiksschattingen wanneer metagegevens over het gebruik van de provider ontbreken## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- Er zijn synchronisatiefouten opgetreden, maar de lokale runtime gaat door -- Scheduler heeft logica die geschikt is voor opnieuw proberen, maar periodieke uitvoering roept momenteel standaard synchronisatie met één poging aan## 5) Data Integrity +## 3) Stream Safety -- SQLite-schemamigraties en automatische upgrade-hooks bij het opstarten -- verouderd JSON → SQLite-migratiecompatibiliteitspad## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Bronnen voor runtime-zichtbaarheid: +## 4) Cloud Sync Degradation -- consolelogboeken van `src/sse/utils/logger.ts` -- gebruiksaggregaten per verzoek in SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- Gedetailleerde payload-opnamen in vier fasen in SQLite (`request_detail_logs`) wanneer `settings.detailed_logs_enabled=true` -- tekstueel verzoek status log in `log.txt` (optioneel/compatibel) -- optionele diepe verzoek-/vertaallogboeken onder `logs/` wanneer `ENABLE_REQUEST_LOGS=true` -- Eindpunten voor dashboardgebruik (`/api/usage/*`) voor UI-gebruik +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -Bij het gedetailleerd vastleggen van de payload van verzoeken worden maximaal vier JSON-payloadfasen per gerouteerde oproep opgeslagen: +## 5) Data Integrity -- ruwe aanvraag ontvangen van de klant -- vertaald verzoek dat daadwerkelijk stroomopwaarts is verzonden -- respons van de provider gereconstrueerd als JSON; gestreamde antwoorden worden gecomprimeerd tot de uiteindelijke samenvatting plus stream-metagegevens -- definitieve klantreactie geretourneerd door OmniRoute; gestreamde antwoorden worden opgeslagen in hetzelfde compacte samenvattingsformulier## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- JWT-geheim (`JWT_SECRET`) beveiligt de verificatie/ondertekening van dashboardsessiecookies -- Initiële wachtwoord-bootstrap (`INITIAL_PASSWORD`) moet expliciet worden geconfigureerd voor inrichting bij eerste uitvoering -- API-sleutel HMAC-geheim (`API_KEY_SECRET`) beveiligt het gegenereerde lokale API-sleutelformaat -- Providergeheimen (API-sleutels/tokens) worden bewaard in de lokale database en moeten worden beschermd op bestandssysteemniveau -- Cloudsynchronisatie-eindpunten zijn afhankelijk van API-sleutelauthenticatie en machine-ID-semantiek## Environment and Runtime Matrix +## Observability and Operational Signals -Omgevingsvariabelen die actief worden gebruikt door code: +Runtime visibility sources: + +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption + +Detailed request payload capture stores up to four JSON payload stages per routed call: + +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: - App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` -- Opslag: `DATA_DIR` -- Compatibel knooppuntgedrag: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Optionele overschrijving van de opslagbasis (Linux/macOS wanneer `DATA_DIR` niet is ingesteld): `XDG_CONFIG_HOME` -- Beveiligingshashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Logboekregistratie: `ENABLE_REQUEST_LOGS` -- Synchroniseren/cloud-URL's: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- Uitgaande proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` en varianten in kleine letters -- SOCKS5-functievlaggen: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- Platform-/runtime-helpers (niet app-specifieke configuratie): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` -1. `usageDb` en `localDb` delen hetzelfde basisdirectorybeleid (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) met oudere bestandsmigratie. -2. `/api/v1/route.ts` delegeert naar dezelfde uniforme catalogusbouwer die wordt gebruikt door `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) om semantische drift te voorkomen. -3. Verzoeklogger schrijft volledige headers/body indien ingeschakeld; behandel de logmap als gevoelig. -4. Het gedrag van de cloud is afhankelijk van de juiste `NEXT_PUBLIC_BASE_URL` en de bereikbaarheid van het cloudeindpunt. -5. De map `open-sse/` wordt gepubliceerd als het `@omniroute/open-sse`**npm-werkruimtepakket**. De broncode importeert het via `@omniroute/open-sse/...` (opgelost door Next.js `transpilePackages`). Bestandspaden in dit document gebruiken nog steeds de mapnaam `open-sse/` voor consistentie. -6. Grafieken in het dashboard maken gebruik van**Recharts**(op SVG-basis) voor toegankelijke, interactieve analytische visualisaties (staafdiagrammen voor modelgebruik, uitsplitsingstabellen van providers met succespercentages). -7. E2E-tests gebruiken**Playwright**(`tests/e2e/`), uitgevoerd via `npm run test:e2e`. Unit-tests gebruiken**Node.js test runner**(`tests/unit/`), uitgevoerd via `npm run test:unit`. De broncode onder `src/` is**TypeScript**(`.ts`/`.tsx`); de `open-sse/` werkruimte blijft JavaScript (`.js`). -8. De instellingenpagina is onderverdeeld in 5 tabbladen: Beveiliging, Routing (6 globale strategieën: eerst vullen, round-robin, p2c, willekeurig, minst gebruikt, kostengeoptimaliseerd), veerkracht (bewerkbare snelheidslimieten, stroomonderbreker, beleid), AI (denkbudget, systeemprompt, promptcache), Geavanceerd (proxy).## Operational Verification Checklist +## Known Architectural Notes -- Bouw vanaf de bron: `npm run build` -- Bouw Docker-image: `docker build -t omniroute .` -- Start de service en controleer: -- `GET /api/instellingen` -- `KRIJG /api/v1/models` -- CLI-doelbasis-URL moet `http://:20128/v1` zijn wanneer `PORT=20128` +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` +- `GET /api/v1/models` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/nl/docs/FEATURES.md b/docs/i18n/nl/docs/FEATURES.md index c7b263249e..7448db401c 100644 --- a/docs/i18n/nl/docs/FEATURES.md +++ b/docs/i18n/nl/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Visuele gids voor elke sectie van het OmniRoute-dashboard.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Beheer AI-providerverbindingen: OAuth-providers (Claude Code, Codex, Gemini CLI), API-sleutelproviders (Groq, DeepSeek, OpenRouter) en gratis providers (Qoder, Qwen, Kiro). Kiro-accounts omvatten het bijhouden van het kredietsaldo: resterende tegoeden, totaalbedrag en verlengingsdatum zichtbaar in Dashboard → Gebruik.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Creëer modelrouteringscombinaties met 6 strategieën: prioriteit, gewogen, round-robin, willekeurig, minst gebruikt en kostengeoptimaliseerd. Elke combo koppelt meerdere modellen met automatische fallback en bevat snelle sjablonen en gereedheidscontroles.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Uitgebreide gebruiksanalyses met tokenverbruik, kostenramingen, activiteiten-heatmaps, wekelijkse distributiegrafieken en uitsplitsingen per provider.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Realtime monitoring: uptime, geheugen, versie, latentiepercentielen (p50/p95/p99), cachestatistieken en status van stroomonderbrekers van de provider.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Vier modi voor het debuggen van API-vertalingen:**Playground**(formaatconverter),**Chat Tester**(live verzoeken),**Test Bench**(batchtests) en**Live Monitor**(realtime stream).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Test elk model rechtstreeks vanaf het dashboard. Selecteer provider, model en eindpunt, schrijf prompts met Monaco Editor, stream reacties in realtime, beëindig halverwege de stream en bekijk timingstatistieken.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Aanpasbare kleurthema's voor het hele dashboard. Kies uit 7 vooraf ingestelde kleuren (koraal, blauw, rood, groen, violet, oranje, cyaan) of creëer een aangepast thema door een willekeurige hex-kleur te kiezen. Ondersteunt lichte, donkere en systeemmodus.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Uitgebreid instellingenpaneel met tabbladen: +Comprehensive settings panel with tabs: --**Algemeen**— Systeemopslag, back-upbeheer (database exporteren/importeren) -**Uiterlijk**: themakiezer (donker/licht/systeem), voorinstellingen voor kleurthema's en aangepaste kleuren, zichtbaarheid van gezondheidslogboeken, bedieningselementen voor zichtbaarheid van items in de zijbalk -**Beveiliging**— API-eindpuntbescherming, aangepaste providerblokkering, IP-filtering, sessie-informatie -**Routing**— Modelaliassen, verslechtering van achtergrondtaak -**Veerkracht**— persistentie van tarieflimieten, afstemming van stroomonderbrekers, automatische uitschakeling van geblokkeerde accounts, monitoring van de vervaldatum van de provider -**Geavanceerd**— Configuratieoverschrijvingen, configuratie-audittraject, terugvaldegradatiemodus![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Configuratie met één klik voor AI-coderingstools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor en Factory Droid. Beschikt over geautomatiseerde configuratie-toepas/reset, verbindingsprofielen en modeltoewijzing.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Dashboard voor het ontdekken en beheren van CLI-agents. Toont een raster van 14 ingebouwde agenten (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) met: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Installatiestatus**— Geïnstalleerd/niet gevonden met versiedetectie -**Protocolbadges**— stdio, HTTP, etc. -**Aangepaste agenten**— Registreer elke CLI-tool via een formulier (naam, binair bestand, versieopdracht, spawn-args) -**CLI Fingerprint Matching**— Schakel per provider om de handtekeningen van native CLI-verzoeken te matchen, waardoor het verbodsrisico wordt verminderd terwijl het proxy-IP behouden blijft--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Genereer afbeeldingen, video's en muziek vanaf het dashboard. Ondersteunt OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open en MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Realtime logboekregistratie van verzoeken met filtering op provider, model, account en API-sleutel. Toont statuscodes, tokengebruik, latentie en responsdetails.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Uw uniforme API-eindpunt met uitsplitsing van de mogelijkheden: chatvoltooiingen, respons-API, insluitingen, het genereren van afbeeldingen, herrangschikking, audiotranscriptie, tekst-naar-spraak, moderaties en geregistreerde API-sleutels. Cloudflare Quick Tunnel-integratie en cloudproxy-ondersteuning voor externe toegang.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -API-sleutels maken, bereiken en intrekken. Elke sleutel kan worden beperkt tot specifieke modellen/providers met volledige toegang of alleen-lezen-rechten. Visueel sleutelbeheer met gebruiksregistratie.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Administratieve actietracking met filtering op actietype, actor, doel, IP-adres en tijdstempel. Volledige geschiedenis van beveiligingsgebeurtenissen.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Native Electron desktop-app voor Windows, macOS en Linux. Voer OmniRoute uit als een zelfstandige toepassing met systeemvakintegratie, offline ondersteuning, automatische updates en installatie met één klik. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Belangrijkste kenmerken: +Key features: -- Polling van servergereedheid (geen leeg scherm bij koude start) -- Systeemvak met poortbeheer -- Inhoudsbeveiligingsbeleid -- Vergrendeling met één exemplaar -- Automatische update bij opnieuw opstarten -- Platform-voorwaardelijke gebruikersinterface (macOS-verkeerslichten, standaardtitelbalk van Windows/Linux) -- Hardened Electron build-verpakking - symlinked `node_modules` in de stand-alone bundel wordt gedetecteerd en afgewezen vóór het verpakken, waardoor runtime-afhankelijkheid van de build-machine wordt voorkomen (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Zie [`electron/README.md`](../electron/README.md) voor volledige documentatie. +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/nl/docs/TROUBLESHOOTING.md b/docs/i18n/nl/docs/TROUBLESHOOTING.md index 37f7cd51f1..15dc7ab895 100644 --- a/docs/i18n/nl/docs/TROUBLESHOOTING.md +++ b/docs/i18n/nl/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -Veelvoorkomende problemen en oplossingen voor OmniRoute.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Probleem | Oplossing | -| ----------------------------------- | --------------------------------------------------------------------------- | --- | -| Eerste login werkt niet | Stel `INITIAL_PASSWORD` in `.env` in (geen hardgecodeerde standaard) | -| Dashboard opent op verkeerde poort | Stel `PORT=20128` en `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Geen verzoeklogboeken onder `logs/` | Stel `ENABLE_REQUEST_LOGS=true` | in | -| EACCES: toestemming geweigerd | Stel `DATA_DIR=/path/to/writable/dir` in om `~/.omniroute` te overschrijven | -| Routeringsstrategie bespaart niet | Update naar v1.4.11+ (Zod-schemafix voor persistentie van instellingen) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Oorzaak:**Providerquotum is opgebruikt. +**Cause:** Provider quota exhausted. -**Opgelost:** +**Fix:** -1. Controleer de dashboardquotatracker -2. Gebruik een combo met fallback-lagen -3. Schakel over naar het goedkopere/gratis niveau### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Oorzaak:**Abonnementsquota zijn opgebruikt. +### Rate Limiting -**Opgelost:** +**Cause:** Subscription quota exhausted. -- Terugval toevoegen: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Gebruik GLM/MiniMax als goedkope back-up### OAuth Token Expired +**Fix:** -OmniRoute vernieuwt tokens automatisch. Als de problemen aanhouden: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Dashboard → Provider → Opnieuw verbinden -2. Verwijder de providerverbinding en voeg deze opnieuw toe--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Controleer of `BASE_URL` verwijst naar uw actieve exemplaar (bijvoorbeeld `http://localhost:20128`) -2. Controleer of `CLOUD_URL` verwijst naar uw cloudeindpunt (bijvoorbeeld `https://omniroute.dev`) -3. Zorg ervoor dat de waarden van `NEXT_PUBLIC_*` uitgelijnd zijn met de waarden op de server### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Symptoom:**`Onverwacht token 'd'...` op cloudeindpunt voor niet-streaming oproepen. +### Cloud `stream=false` Returns 500 -**Oorzaak:**Upstream retourneert SSE-payload terwijl de client JSON verwacht. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Oplossing:**Gebruik `stream=true` voor directe cloudoproepen. Lokale runtime omvat SSE → JSON-fallback.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Maak een nieuwe sleutel vanuit het lokale dashboard (`/api/keys`) -2. Voer cloudsynchronisatie uit: Schakel Cloud in → Nu synchroniseren -3. Oude/niet-gesynchroniseerde sleutels kunnen nog steeds '401' retourneren in de cloud--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Controleer runtimevelden: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. Voor draagbare modus: gebruik afbeeldingsdoel `runner-cli` (gebundelde CLI's) -3. Voor host-aankoppelmodus: stel `CLI_EXTRA_PATHS` in en koppel de hostbin-directory aan als alleen-lezen -4. Indien `geïnstalleerd=true` en `runnable=false`: binair bestand is gevonden maar de gezondheidscontrole is mislukt### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Controleer gebruiksstatistieken in Dashboard → Gebruik -2. Schakel het primaire model over naar GLM/MiniMax -3. Gebruik de gratis laag (Gemini CLI, Qoder) voor niet-kritieke taken -4. Stel kostenbudgetten per API-sleutel in: Dashboard → API-sleutels → Budget--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Stel `ENABLE_REQUEST_LOGS=true` in uw `.env`-bestand in. Logboeken verschijnen onder de map `logs/`.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Hoofdstatus: `${DATA_DIR}/storage.sqlite` (providers, combo's, aliassen, sleutels, instellingen) -- Gebruik: SQLite-tabellen in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optioneel `${DATA_DIR}/log.txt` en `${DATA_DIR}/call_logs/` -- Logboeken aanvragen: `/logs/...` (wanneer `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Wanneer de stroomonderbreker van een provider OPEN is, worden verzoeken geblokkeerd totdat de cooldown is verstreken. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Opgelost:** +**Fix:** -1. Ga naar**Dashboard → Instellingen → Veerkracht** -2. Controleer de stroomonderbrekerkaart van de betreffende provider -3. Klik op**Alles resetten**om alle onderbrekers te wissen, of wacht tot de cooldown is verstreken -4. Controleer of de provider daadwerkelijk beschikbaar is voordat u reset### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Als een aanbieder herhaaldelijk in de OPEN-status komt: +### Provider keeps tripping the circuit breaker -1. Controleer**Dashboard → Gezondheid → Providergezondheid**voor het foutpatroon -2. Ga naar**Instellingen → Veerkracht → Providerprofielen**en verhoog de foutdrempel -3. Controleer of de provider de API-limieten heeft gewijzigd of herauthenticatie vereist -4. Controleer latentie-telemetrie: hoge latentie kan op time-outs gebaseerde fouten veroorzaken--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Zorg ervoor dat u het juiste voorvoegsel gebruikt: `deepgram/nova-3` of `assemblyai/best` -- Controleer of de provider is verbonden in**Dashboard → Providers**### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Controleer ondersteunde audioformaten: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- Controleer of de bestandsgrootte binnen de limieten van de provider ligt (doorgaans < 25 MB) -- Controleer de geldigheid van de API-sleutel van de provider op de providerkaart--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Gebruik**Dashboard → Vertaler**om problemen met de vertaling van formaten op te lossen: +Use **Dashboard → Translator** to debug format translation issues: -| Modus | Wanneer gebruiken | -| --------------- | ---------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Speeltuin** | Vergelijk invoer-/uitvoerformaten naast elkaar - plak een mislukt verzoek om te zien hoe het zich vertaalt | -| **Chattester** | Verzend live berichten en inspecteer de volledige payload van verzoeken/antwoorden, inclusief headers | -| **Proefbank** | Voer batchtests uit voor indelingscombinaties om te ontdekken welke vertalingen niet werken | -| **Livemonitor** | Bekijk de realtime aanvraagstroom om intermitterende vertaalproblemen op te sporen | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Thinking-tags verschijnen niet**— Controleer of de doelaanbieder het denken en de instelling van het denkbudget ondersteunt -**Tooloproepen vervallen**— Bij sommige formaatvertalingen kunnen niet-ondersteunde velden worden verwijderd; verifiëren in Speeltuinmodus -**Systeemprompt ontbreekt**— Claude en Gemini behandelen de systeemprompts anders; controleer de vertalingsuitvoer -**SDK retourneert onbewerkte tekenreeks in plaats van object**— Opgelost in v1.1.0: respons sanitizer verwijdert nu niet-standaard velden (`x_groq`, `usage_breakdown`, enz.) die OpenAI SDK Pydantic-validatiefouten veroorzaken -**GLM/ERNIE wijst de `systeem`-rol af**— Opgelost in v1.1.0: de rolnormalizer voegt systeemberichten automatisch samen met gebruikersberichten voor incompatibele modellen -**`rol van ontwikkelaar` wordt niet herkend**— Opgelost in v1.1.0: automatisch geconverteerd naar `systeem` voor niet-OpenAI-providers -**`json_schema` werkt niet met Gemini**— Opgelost in v1.1.0: `response_format` wordt nu geconverteerd naar Gemini's `responseMimeType` + `responseSchema`--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- Automatische tarieflimiet is alleen van toepassing op API-sleutelproviders (niet op OAuth/abonnement) -- Controleer of bij Instellingen → Veerkracht → Providerprofielen\*\*automatische tarieflimiet is ingeschakeld -- Controleer of de provider statuscodes '429' of headers 'Retry-After' retourneert### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -Providerprofielen ondersteunen deze instellingen: +### Tuning exponential backoff --**Basisvertraging**— Initiële wachttijd na eerste storing (standaard: 1s) -**Max. vertraging**— Maximale wachttijdlimiet (standaard: 30s) -**Vermenigvuldiger**— Hoeveel vertraging per opeenvolgende fout moet worden vergroot (standaard: 2x)### Anti-thundering herd +Provider profiles support these settings: -Wanneer veel gelijktijdige verzoeken een provider met een beperkte snelheid bereiken, gebruikt OmniRoute mutex + automatische snelheidsbeperking om verzoeken te serialiseren en trapsgewijze fouten te voorkomen. Dit gebeurt automatisch voor API-sleutelproviders.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Sommige OmniRoute-gebruikers plaatsen de gateway vóór RAG- of agentstacks. In deze instellingen is het gebruikelijk om een ​​vreemd patroon te zien: OmniRoute ziet er gezond uit (providers actief, routeringsprofielen ok, geen waarschuwingen over de snelheidslimiet), maar het uiteindelijke antwoord is nog steeds verkeerd. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -In de praktijk komen deze incidenten meestal van de stroomafwaartse RAG-pijpleiding en niet van de gateway zelf. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Als u een gedeelde woordenschat wilt om deze fouten te beschrijven, kunt u de WFGY ProblemMap gebruiken, een externe MIT-licentietekstbron die zestien terugkerende RAG / LLM-foutpatronen definieert. Op een hoog niveau omvat het: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- retrieval drift en verbroken contextgrenzen -- lege of verouderde indexen en vectorwinkels -- inbedding versus semantische mismatch -- problemen met snelle montage en contextvensters -- logische ineenstorting en overmoedige antwoorden -- mislukkingen in de lange keten en de coördinatie van agenten -- Multi-agentgeheugen en rolafwijking -- implementatie- en bootstrap-bestellingsproblemen +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -Het idee is simpel: +The idea is simple: -1. Wanneer je een slechte reactie onderzoekt, leg dan vast: - - gebruikerstaak en -verzoek - - route- of providercombinatie in OmniRoute - - elke RAG-context die stroomafwaarts wordt gebruikt (opgehaalde documenten, tooloproepen, enz.) -2. Wijs het incident toe aan een of twee WFGY ProblemMap-nummers (`No.1` … `No.16`). -3. Bewaar het nummer in uw eigen dashboard, runbook of incidenttracker naast de OmniRoute-logboeken. -4. Gebruik de bijbehorende WFGY-pagina om te beslissen of u uw RAG-stack, retriever of routeringsstrategie moet wijzigen. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -Volledige tekst en concrete recepten staan hier (MIT-licentie, alleen tekst): +Full text and concrete recipes live here (MIT license, text only): [WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -U kunt deze sectie negeren als u geen RAG- of agentpijplijnen achter OmniRoute uitvoert.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**GitHub-problemen**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Architectuur**: zie [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) voor interne details -**API-referentie**: zie [`docs/API_REFERENCE.md`](API_REFERENCE.md) voor alle eindpunten -**Gezondheidsdashboard**: controleer**Dashboard → Gezondheid**voor de realtime systeemstatus -**Vertaler**: gebruik**Dashboard → Vertaler**om formaatproblemen op te lossen +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/nl/llm.txt b/docs/i18n/nl/llm.txt new file mode 100644 index 0000000000..d8a3734271 --- /dev/null +++ b/docs/i18n/nl/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Nederlands) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Overzicht + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Beveiliging +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/no/README.md b/docs/i18n/no/README.md index d9f6d93255..b61342156b 100644 --- a/docs/i18n/no/README.md +++ b/docs/i18n/no/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Din universelle API-proxy – ett endepunkt, 60+ leverandører, null nedetid. Nå med**MCP Server (25 verktøy)**,**A2A Protocol**,**Memory/Skills Systems**og**Electron Desktop App**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Chatfullføringer • Innebygginger • Bildegenerering • Video • Musikk • Lyd • Rerangering •**Nettsøk**• MCP-server • A2A-protokoll • 100 % TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Din universelle API-proxy – ett endepunkt, 60+ leverandører, null nedetid. N [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Nettsted](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Funksjoner](#-key-features) • [📖 Dokumenter](#-dokumentasjon) • [💰 Priser](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Tilgjengelig på:**🇺🇸 [engelsk](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [norsk](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [filippinsk](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,28 +60,30 @@ _Din universelle API-proxy – ett endepunkt, 60+ leverandører, null nedetid. N ## 📸 Dashboard Preview - -Klikk for å se skjermbilder av oversikten +
+Click to see dashboard screenshots -| Side | Skjermbilde | -| ----------------- | ------------------------------------------------- | ---------- | -| **Tilbydere** | ![Providers](docs/screenshots/01-providers.png) | -| **Komboer** | ![Combos](docs/screenshots/02-combos.png) | -| **Analyse** | ![Analytics](docs/screenshots/03-analytics.png) | -| **Helse** | ![Helse](docs/screenshots/04-health.png) | -| **Oversetter** | ![Translator](docs/screenshots/05-translator.png) | -| **Innstillinger** | ![Settings](docs/screenshots/06-settings.png) | -| **CLI-verktøy** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | -| **Brukslogger** | ![Usage](docs/screenshots/08-usage.png) | -| **Endepunkter** | ![Endpoints](docs/screenshots/09-endpoint.png) |
| +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | + + --- ### 🤖 Free AI Provider for your favorite coding agents -_Koble til ethvert AI-drevet IDE- eller CLI-verktøy gjennom OmniRoute – gratis API-gateway for ubegrenset koding._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ - +
@@ -88,21 +97,21 @@ _Koble til ethvert AI-drevet IDE- eller CLI-verktøy gjennom OmniRoute – grati NanoBot
NanoBot

- ⭐ 20,9K + ⭐ 20.9K
PicoClaw
PicoClaw

- ⭐ 14,6K + ⭐ 14.6K
ZeroClaw
ZeroClaw

- ⭐ 9,9K + ⭐ 9.9K
@@ -125,481 +134,555 @@ _Koble til ethvert AI-drevet IDE- eller CLI-verktøy gjennom OmniRoute – grati Codex CLI
Codex CLI

- ⭐ 60,8K + ⭐ 60.8K
Claude Code
Claude Code

- ⭐ 67,3K + ⭐ 67.3K
Gemini CLI
Gemini CLI

- ⭐ 94,7K + ⭐ 94.7K
- Kilokode
- Kilokode + Kilo Code
+ Kilo Code

⭐ 15.5K
-📡 Alle agenter kobler til via http://localhost:20128/v1 eller http://cloud.omniroute.online/v1 – én konfigurasjon, ubegrenset med modeller og kvote--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Slutt å kaste bort penger og nå grensene:** +**Stop wasting money and hitting limits:** -- Abonnementskvoten utløper ubrukt hver måned -- Satsgrenser hindrer deg i midtkoding -- Dyre APIer ($20–50/måned per leverandør) -- Manuell bytting mellom leverandører +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute løser dette:** +**OmniRoute solves this:** -- ✅**Maksimer abonnementer**- Spor kvote, bruk hver bit før tilbakestilling -- ✅**Automatisk fallback**- Abonnement → API-nøkkel → Billig → Gratis, null nedetid -- ✅**Multi-konto**- Round-robin mellom kontoer per leverandør -- ✅**Universal**- Fungerer med Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, hvilket som helst CLI-verktøy--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Bli med i fellesskapet vårt!**[WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Få hjelp, del tips og hold deg oppdatert. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Nettsted**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Problemer**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [fellesskapsgruppe](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Bidra**: Se [CONTRIBUTING.md](CONTRIBUTING.md), åpne en PR, eller velg en "god første utgave". -**Originalt prosjekt**: [9router av decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Når du åpner et problem, kjør systeminfo-kommandoen og legg ved den genererte filen:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Dette genererer en `system-info.txt` med Node.js-versjonen din, OmniRoute-versjonen, OS-detaljer, installerte CLI-verktøy (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2-status og systempakker – alt vi trenger for å gjenskape problemet raskt. Legg ved filen direkte til GitHub-problemet ditt.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Hver utviklere som bruker AI-verktøy møter disse problemene daglig.**OmniRoute ble bygget for å løse dem alle – fra kostnadsoverskridelser til regionale blokker, fra ødelagte OAuth-flyter til protokolloperasjoner og observerbarhet i bedrifter. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. - -💸 1. "Jeg betaler for et dyrt abonnement, men blir fortsatt avbrutt av limits" +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Utviklere betaler $20–200/måned for Claude Pro, Codex Pro eller GitHub Copilot. Selv om du betaler, har kvoten et tak – 5 timers bruk, ukentlige grenser eller rategrenser per minutt. Midtkodingsøkt, leverandøren slutter å svare og utvikleren mister flyt og produktivitet. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Hvordan OmniRoute løser det:** +**How OmniRoute solves it:** --**Smart 4-lags fallback**— Hvis abonnementskvoten går tom, omdirigeres automatisk til API-nøkkel → Billig → Gratis med null manuell intervensjon --**Sporing av leverandørgrenser**— Bufret kvoteøyeblikksbilder oppdateres på en serversideplan (standard `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) med manuell oppdatering tilgjengelig i brukergrensesnittet --**Støtte for flere kontoer**- Flere kontoer per leverandør med automatisk round-robin - når en går tom, bytter du til den neste --**Egendefinerte kombinasjoner**— Tilpassbare reservekjeder med 9 balanseringsstrategier (prioritet, vektet, fyll først, round-robin, P2C, tilfeldig, minst brukt, kostnadsoptimalisert, strengt tilfeldig) --**Codex Business Quotas**— Overvåking av bedrifts-/teamarbeidsområdekvoter direkte i dashbordet
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard - -🔌 2. "Jeg trenger å bruke flere leverandører, men hver har en annen API" + -OpenAI bruker ett format, Claude (Anthropic) bruker et annet, Gemini enda et annet. Hvis en utvikler ønsker å teste modeller fra forskjellige leverandører eller fallback mellom dem, må de rekonfigurere SDK-er, endre endepunkter, håndtere inkompatible formater. Tilpassede leverandører (FriendLI, NIM) har ikke-standardmodellende endepunkter. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Hvordan OmniRoute løser det:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Unified Endpoint**- En enkelt "http://localhost:20128/v1" fungerer som proxy for alle 60+ leverandører --**Formatoversettelse**— Automatisk og gjennomsiktig: OpenAI ↔ Claude ↔ Gemini ↔ Responses API -–**Responsrensing**– Fjerner ikke-standardiserte felt (`x_groq`, `usage_breakdown`, `service_tier`) som bryter OpenAI SDK v1.83+ --**Rollenormalisering**— Konverterer `utvikler` → `system` for ikke-OpenAI-leverandører; `system` → `bruker` for GLM/ERNIE --**Think Tag Extraction**— Trekker ut «»-blokker fra modeller som DeepSeek R1 til standardisert «reasoning_content» --**Structured Output for Gemini**— `json_schema` → `responseMimeType`/`responseSchema` automatisk konvertering --**`stream` er standard til "false"**- Justerer med OpenAI-spesifikasjoner, og unngår uventet SSE i Python/Rust/Go SDK-er
+**How OmniRoute solves it:** - -🌐 3. «Min AI-leverandør blokkerer min region/land» +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Leverandører som OpenAI/Codex blokkerer tilgang fra visse geografiske områder. Brukere får feil som «unsupported_country_region_territory» under OAuth- og API-tilkoblinger. Dette er spesielt frustrerende for utviklere fra utviklingsland. + -**Hvordan OmniRoute løser det:** +
+🌐 3. "My AI provider blocks my region/country" --**3-Level Proxy Config**— Konfigurerbar proxy på 3 nivåer: global (all trafikk), per leverandør (kun én leverandør) og per tilkobling/nøkkel --**Fargekodede proxy-merker**— Visuelle indikatorer: 🟢 global proxy, 🟡 leverandørproxy, 🔵 tilkoblings proxy, viser alltid IP --**OAuth-tokenutveksling gjennom proxy**- OAuth-flyt går også gjennom proxyen, og løser «unsupported_country_region_territory». --**Test av tilkobling via proxy**— Tilkoblingstester bruker den konfigurerte proxyen (ikke mer direkte forbikobling) --**SOCKS5-støtte**— Full SOCKS5-proxystøtte for utgående ruting --**TLS-fingeravtrykkspoofing**- Nettleserlignende TLS-fingeravtrykk via 'wreq-js' for å omgå botdeteksjon --**🔏 Matching av CLI-fingeravtrykk**— Omorganiserer overskrifter og brødtekstfelter for å matche native CLI-binære signaturer, noe som reduserer risikoen for kontoflagging drastisk. Proxy-IP-en er bevart – du får både skjult**og**IP-maskering samtidig
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. - -🆓 4. "Jeg vil bruke AI for koding, men jeg har ingen penger" +**How OmniRoute solves it:** -Ikke alle kan betale $20–200 per måned for AI-abonnementer. Studenter, utviklere fra fremvoksende land, hobbyfolk og frilansere trenger tilgang til kvalitetsmodeller uten kostnad. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Hvordan OmniRoute løser det:** + --**Gratis-tilbydere innebygd**— Innebygd støtte for 100 % gratis leverandører: Qoder (5 ubegrensede modeller via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited-modeller:-r-modeller:-r qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID gratis), Gemini CLI (180K tokens/måned gratis) --**Ollama Cloud**— Ollama-modeller som er vert for skyen på `api.ollama.com` med gratis "Lett bruk"-lag; bruk `ollamacloud/` prefiks --**Kun gratis kombinasjoner**— Kjede `gc/gemini-3-flash → if/kimi-k2-tenking → qw/qwen3-coder-plus` = $0/måned med null nedetid --**NVIDIA NIM Free Access**— ~40 RPM dev-forever gratis tilgang til 70+ modeller på build.nvidia.com (overgang fra kreditter til rene rategrenser) --**Kostnadsoptimalisert strategi**— Rutingstrategi som automatisk velger den billigste tilgjengelige leverandøren +
+🆓 4. "I want to use AI for coding but I have no money" - -🔒 5. «Jeg må beskytte AI-gatewayen min mot uautorisert tilgang» +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -Når du eksponerer en AI-gateway til nettverket (LAN, VPS, Docker), kan alle med adressen konsumere utviklerens tokens/kvote. Uten beskyttelse er API-er sårbare for misbruk, umiddelbar injeksjon og misbruk. +**How OmniRoute solves it:** -**Hvordan OmniRoute løser det:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**API Key Management**— Generering, rotasjon og scoping per leverandør med en dedikert `/dashboard/api-manager`-side --**Tillatelser på modellnivå**— Begrens API-nøkler til spesifikke modeller ('openai/*', jokertegnmønstre), med Tillat alt/begrens-bryteren --**API Endpoint Protection**— Krev en nøkkel for `/v1/modeller` og blokker spesifikke leverandører fra oppføringen --**Auth Guard + CSRF Protection**— Alle dashbordruter beskyttet med "withAuth" mellomvare + CSRF-tokens --**Rate Limiter**— Per-IP ratebegrensning med konfigurerbare vinduer --**IP-filtrering**— Tillatelsesliste/blokkeringsliste for tilgangskontroll --**Prompt Injection Guard**— Sanitisering mot ondsinnede spørsmålsmønstre --**AES-256-GCM-kryptering**— Legitimasjon kryptert i hvile
+ - -🛑 6. «Tilbyderen min gikk ned og jeg mistet kodeflyten min» +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -AI-leverandører kan bli ustabile, returnere 5xx-feil eller nå midlertidige hastighetsgrenser. Hvis en utvikler er avhengig av en enkelt leverandør, blir de avbrutt. Uten strømbrytere kan gjentatte forsøk krasje applikasjonen. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Hvordan OmniRoute løser det:** +**How OmniRoute solves it:** --**Circuit Breaker per modell**— Automatisk åpning/lukking med konfigurerbare terskler og nedkjøling (Lukket/Åpen/Halv-Åpen), omfang per modell for å unngå kaskadeblokker --**Eksponentiell backoff**— Progressive forsinkelser på nytt forsøk --**Anti-tordenflokk**— Mutex + semaforbeskyttelse mot samtidige stormer på nytt forsøk --**Combo Fallback Chains**— Hvis primærleverandøren mislykkes, faller den automatisk gjennom kjeden uten inngrep --**Combo Circuit Breaker**- Deaktiverer sviktende leverandører automatisk i en kombinasjonskjede --**Helsedashbord**— Oppetidsovervåking, strømbrytertilstander, sperringer, cachestatistikk, p50/p95/p99 latency
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest - -🔧 7. "Å konfigurere hvert AI-verktøy er kjedelig og repeterende" + -Utviklere bruker Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Hvert verktøy trenger en annen konfigurasjon (API-endepunkt, nøkkel, modell). Å konfigurere på nytt når du bytter leverandør eller modell er bortkastet tid. +
+🛑 6. "My provider went down and I lost my coding flow" -**Hvordan OmniRoute løser det:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**CLI Tools Dashboard**— Dedikert side med ett-klikksoppsett for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline --**GitHub Copilot Config Generator**— Genererer `chatLanguageModels.json` for VS-kode med bulk modellvalg --**Onboarding Wizard**— Veiledet 4-trinns oppsett for førstegangsbrukere --**Ett endepunkt, alle modeller**— Konfigurer `http://localhost:20128/v1` én gang, få tilgang til 60+ leverandører
+**How OmniRoute solves it:** - -🔑 8. "Å administrere OAuth-tokens fra flere leverandører er et helvete" +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot – alle bruker OAuth 2.0 med tokens som utløper. Utviklere må re-autentisere hele tiden, håndtere "client_secret is missing", "redirect_uri_mismatch" og feil på eksterne servere. OAuth på LAN/VPS er spesielt problematisk. + -**Hvordan OmniRoute løser det:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Automatisk oppdatering av token**— OAuth-tokener oppdateres i bakgrunnen før utløp --**OAuth 2.0 (PKCE) innebygd**— Automatisk flyt for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder --**Multi-Account OAuth**- Flere kontoer per leverandør via JWT/ID-tokenutvinning --**OAuth LAN/Remote Fix**- Privat IP-deteksjon for `redirect_uri` + manuell URL-modus for eksterne servere --**OAuth Behind Nginx**- Bruker `window.location.origin` for omvendt proxy-kompatibilitet --**Remote OAuth Guide**— Trinn-for-trinn-veiledning for Google Cloud-legitimasjon på VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. - -📊 9. "Jeg vet ikke hvor mye jeg bruker eller hvor" +**How OmniRoute solves it:** -Utviklere bruker flere betalte leverandører, men har ikke noe enhetlig syn på utgifter. Hver leverandør har sitt eget faktureringsdashbord, men det er ingen konsolidert visning. Uventede kostnader kan hope seg opp. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Hvordan OmniRoute løser det:** + --**Dashboard for kostnadsanalyse**— Kostnadssporing per token og budsjettadministrasjon per leverandør -–**Budsjettgrenser per nivå**– Utgiftstak per nivå som utløser automatisk fallback --**Priskonfigurasjon per modell**— Konfigurerbare priser per modell --**Bruksstatistikk per API-nøkkel**— Antall forespørsler og sist brukte tidsstempel per nøkkel --**Analytics Dashboard**— Statistiske kort, modellbruksdiagram, leverandørtabell med suksessrater og latens +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" - -🐛 10. "Jeg kan ikke diagnostisere feil og problemer i AI-anrop" +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Når et anrop mislykkes, vet ikke utvikleren om det var en takstgrense, utløpt token, feil format eller leverandørfeil. Fragmenterte logger på tvers av forskjellige terminaler. Uten observerbarhet er feilsøking prøving og feiling. +**How OmniRoute solves it:** -**Hvordan OmniRoute løser det:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Unified Logs Dashboard**— 4 faner: Forespørselslogger, proxy-logger, revisjonslogger, konsoll --**Console Log Viewer**— Viser i sanntid i terminalstil med fargekodede nivåer, automatisk rulling, søk, filter --**SQLite Proxy Logger**— Vedvarende logger som overlever serverstarter --**Translator Playground**— 4 feilsøkingsmoduser: Playground (formatoversettelse), Chat Tester (tur-retur), Test Bench (batch), Live Monitor (sanntid) --**Request Telemetri**— p50/p95/p99 latens + X-Request-Id-sporing --**Filbasert logging med rotasjon**— Applogger roterer etter størrelse, oppbevaringsdager og arkivantall; anropsloggartefakter roterer etter oppbevaringsdager og filantall --**System Info Report**— `npm run system-info` genererer `system-info.txt` med hele miljøet ditt (Node-versjon, OmniRoute-versjon, OS, CLI-verktøy, Docker/PM2-status). Legg den ved når du rapporterer problemer for umiddelbar triage.
+ - -🏗️ 11. "Deployering og vedlikehold av gatewayen er kompleks" +
+📊 9. "I don't know how much I'm spending or where" -Installering, konfigurering og vedlikehold av en AI-proxy på tvers av forskjellige miljøer (lokalt, VPS, Docker, sky) er arbeidskrevende. Problemer som hardkodede baner, "EACCES" på kataloger, portkonflikter og plattformkonstruksjoner gir friksjon. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Hvordan OmniRoute løser det:** +**How OmniRoute solves it:** --**npm global installasjon**— `npm install -g omniroute && omniroute` – ferdig --**Docker Multi-Platform**— AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) --**Docker Compose Profiles**— 'base' (ingen CLI-verktøy) og 'cli' (med Claude Code, Codex, OpenClaw) --**Electron Desktop App**— Innebygd app for Windows/macOS/Linux med systemstatusfelt, automatisk start, offline-modus --**Split-Port Mode**— API og Dashboard på separate porter for avanserte scenarier (omvendt proxy, containernettverk) --**Cloud Sync**— Konfigurer synkronisering på tvers av enheter via Cloudflare Workers --**DB-sikkerhetskopier**— Automatisk sikkerhetskopiering, gjenoppretting, eksport og import av alle innstillinger, med `DISABLE_SQLITE_AUTO_BACKUP` for eksternt administrerte sikkerhetskopier
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency - -🌍 12. "Grensesnittet er kun engelsk, og teamet mitt snakker ikke engelsk" + -Lag i ikke-engelsktalende land, spesielt i Latin-Amerika, Asia og Europa, sliter med grensesnitt som kun er på engelsk. Språkbarrierer reduserer bruken og øker konfigurasjonsfeil. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Hvordan OmniRoute løser det:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Dashboard i18n — 30 språk**— Alle 500+ nøkler oversatt, inkludert arabisk, bulgarsk, dansk, tysk, spansk, finsk, fransk, hebraisk, hindi, ungarsk, indonesisk, italiensk, japansk, koreansk, malaysisk, nederlandsk, norsk, polsk, portugisisk (PT/BR), rumensk, russisk, ukrainsk, ukrainsk, kinesisk, engelsk, kinesisk, ukrainsk, kinesisk, ukrainsk, kinesisk, ukrainsk, kinesisk, ukrainsk, kinesisk --**RTL-støtte**— Høyre-til-venstre-støtte for arabisk og hebraisk --**Multi-Language READMEs**- 30 komplette dokumentasjonsoversettelser --**Språkvelger**— Globusikon i overskriften for sanntidsbytte
+**How OmniRoute solves it:** - -🔄 13. «Jeg trenger mer enn chat — jeg trenger innebygginger, bilder, lyd» +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -AI er ikke bare fullføring av chat. Utviklere må generere bilder, transkribere lyd, lage innbygginger for RAG, omrangere dokumenter og moderere innhold. Hver API har et annet endepunkt og format. + -**Hvordan OmniRoute løser det:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Embeddings**— `/v1/embeddings` med 6 leverandører og 9+ modeller --**Image Generation**— `/v1/images/generations` med 10 leverandører og 20+ modeller (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Tekst-til-video**— `/v1/videoer/generasjoner` — ComfyUI (AnimateDiff, SVD) og SD WebUI --**Tekst-til-musikk**— `/v1/musikk/generasjoner` — ComfyUI (Stable Audio Open, MusicGen) --**Lydtranskripsjon**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Tekst-til-tale**— `/v1/audio/tale` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + eksisterende leverandører --**Moderasjoner**— `/v1/moderasjoner` — Innholdssikkerhetssjekker --**Reranking**— `/v1/rerank' — Reranking av dokumentrelevans --**Responses API**— Full `/v1/responses`-støtte for Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. - -🧪 14. "Jeg har ingen måte å teste og sammenligne kvalitet på tvers av modeller" +**How OmniRoute solves it:** -Utviklere ønsker å vite hvilken modell som er best for deres brukssituasjon – kode, oversettelse, resonnement – men det går tregt å sammenligne manuelt. Det finnes ingen integrerte evalueringsverktøy. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**Hvordan OmniRoute løser det:** + --**LLM-evalueringer**— Gyldent sett-testing med 10 forhåndslastede tilfeller som dekker hilsener, matematikk, geografi, kodegenerering, JSON-overholdelse, oversettelse, nedskrivning, sikkerhetsavslag --**4 matchstrategier**- "eksakt", "inneholder", "regex", "tilpasset" (JS-funksjon) --**Translator Playground Test Bench**— Batchtesting med flere innganger og forventede utganger, sammenligning på tvers av leverandører --**Chattetester**— Full rundtur med visuell responsgjengivelse --**Live Monitor**— Sanntidsstrøm av alle forespørsler som strømmer gjennom proxyen +
+🌍 12. "The interface is English-only and my team doesn't speak English" - -📈 15. «Jeg trenger å skalere uten å miste ytelse» +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -Når forespørselsvolumet vokser, genererer de samme spørsmålene dupliserte kostnader uten å bufre. Uten idempotens, dupliserte forespørsler om avfallsbehandling. Satsgrenser per leverandør må respekteres. +**How OmniRoute solves it:** -**Hvordan OmniRoute løser det:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Semantisk hurtigbuffer**— To-lags cache (signatur + semantisk) reduserer kostnader og ventetid --**Request Idempotency**— 5s dedupliseringsvindu for identiske forespørsler --**Deteksjon av hastighetsgrense**- RPM per leverandør, minimum gap og maksimal samtidig sporing --**Redigerbare frekvensgrenser**— Konfigurerbare standardinnstillinger i Innstillinger → Motstandsdyktighet med utholdenhet --**API Key Validation Cache**— 3-lags cache for produksjonsytelse --**Helsedashbord med telemetri**— p50/p95/p99-forsinkelse, hurtigbufferstatistikk, oppetid
+ - -🤖 16. «Jeg vil kontrollere modellatferd globalt» +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Utviklere som vil ha alle svar på et spesifikt språk, med en bestemt tone, eller som ønsker å begrense resonnement-tokens. Å konfigurere dette i hvert verktøy/hver forespørsel er upraktisk. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Hvordan OmniRoute løser det:** +**How OmniRoute solves it:** --**System Prompt Injection**— Global forespørsel brukt på alle forespørsler --**Thinking Budget Validation**— Reasoning token allocation control per request (passthrough, auto, custom, adaptive) --**9 rutingstrategier**— Globale strategier som bestemmer hvordan forespørsler distribueres --**Wildcard Router**— `leverandør/*`-mønstre ruter dynamisk til enhver leverandør --**Kombo aktiver/deaktiver veksle**— Veksle kombinasjoner direkte fra dashbordet --**Tilkobling av leverandør**— Aktiver/deaktiver alle tilkoblinger for en leverandør med ett klikk --**Blokkerte leverandører**— Ekskluder spesifikke leverandører fra `/v1/models`-oppføringen
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex - -🧰 17. «Jeg trenger MCP-verktøy som førsteklasses produktegenskaper» + -Mange AI-gatewayer avslører MCP bare som en skjult implementeringsdetalj. Team trenger et synlig, håndterbart driftslag. +
+🧪 14. "I have no way to test and compare quality across models" -**Hvordan OmniRoute løser det:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP vises i dashbordnavigasjons- og endepunktprotokollfanen -- Dedikert MCP-administrasjonsside med prosess, verktøy, omfang og revisjon -- Innebygd hurtigstart for `omniroute --mcp` og klient onboarding
+**How OmniRoute solves it:** - -🧠 18. "Jeg trenger A2A-orkestrering med synkronisering + strømmeoppgavestier" +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Agentarbeidsflyter trenger både direkte svar og langvarig strømmet utførelse med livssykluskontroll. + -**Hvordan OmniRoute løser det:** +
+📈 15. "I need to scale without losing performance" -- A2A JSON-RPC-endepunkt ('POST /a2a') med 'melding/send' og 'melding/strøm' -- SSE-streaming med forplantning av terminaltilstand -- Oppgavelivssyklus-API-er for "oppgaver/hent" og "oppgaver/avbryt".
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. - -🛰️ 19. «Jeg trenger ekte MCP-prosesshelse, ikke gjettet status» +**How OmniRoute solves it:** -Operasjonelle team må vite om MCP faktisk er i live, ikke bare om en API er tilgjengelig. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**Hvordan OmniRoute løser det:** + -- Runtime hjerteslag-fil med PID, tidsstempler, transport, verktøytelling og omfangsmodus -- MCP status API som kombinerer hjerteslag + nylig aktivitet -- UI-statuskort for prosess/oppetid/hjerteslag +
+🤖 16. "I want to control model behavior globally" - -📋 20. "Jeg trenger reviderbar MCP-verktøykjøring" +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Når verktøy muterer konfigurasjon eller utløser operasjonshandlinger, trenger teamene rettsmedisinsk sporbarhet. +**How OmniRoute solves it:** -**Hvordan OmniRoute løser det:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- SQLite-støttet revisjonslogging for MCP-verktøykall -- Filtrerer etter verktøy, suksess/fiasko, API-nøkkel og paginering -- Dashboard revisjonstabell + statistikkendepunkter for automatisering
+ - -🔐 21. «Jeg trenger scoped MCP-tillatelser per integrasjon» +
+🧰 17. "I need MCP tools as first-class product capabilities" -Ulike klienter bør ha minst privilegert tilgang til verktøykategorier. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Hvordan OmniRoute løser det:** +**How OmniRoute solves it:** -- 10 granulære MCP-skoper for kontrollert verktøytilgang -- Håndhevelse av omfang og synlighet i MCP-administrasjonsgrensesnittet -- Sikker standardstilling for operativt verktøy
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding - -⚙️ 22. «Jeg trenger driftskontroller uten å omdistribuere» + -Lag trenger raske endringer i kjøretiden under hendelser eller kostnadshendelser. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**Hvordan OmniRoute løser det:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Bytt kombinasjonsaktivering direkte fra MCP-dashbordet -- Bruk robusthetsprofiler fra forhåndsdefinerte policypakker -- Tilbakestill strømbryterens tilstand fra samme driftspanel
+**How OmniRoute solves it:** - -🔄 23. «Jeg trenger live A2A-oppgavelivssyklussynlighet og kansellering» +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Uten livssyklussynlighet blir oppgavehendelser vanskelig å triage. + -**Hvordan OmniRoute løser det:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Oppgaveliste/filtrering etter tilstand/ferdighet med paginering -- Drill-down på oppgavemetadata, hendelser og artefakter -- Sluttpunkt for kansellering av oppgave og UI-handling med bekreftelse
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. - -🌊 24. «Jeg trenger aktive strømmemålinger for A2A-last» +**How OmniRoute solves it:** -Strømmearbeidsflyter krever operasjonell innsikt i samtidighet og direkteforbindelser. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**Hvordan OmniRoute løser det:** + -- Aktive strømtellere integrert i A2A-status -- Tidsstempel for siste oppgave og antall per stat -- A2A dashbordkort for operasjonsovervåking i sanntid +
+📋 20. "I need auditable MCP tool execution" - -🪪 25. «Jeg trenger standard agentoppdagelse for klienter» +When tools mutate config or trigger ops actions, teams need forensic traceability. -Eksterne klienter og orkestratorer trenger maskinlesbare metadata for onboarding. +**How OmniRoute solves it:** -**Hvordan OmniRoute løser det:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- Agentkort eksponert på `/.well-known/agent.json` -- Evner og ferdigheter vist i ledelsens brukergrensesnitt -- A2A status API inkluderer oppdagelsesmetadata for automatisering
+ - -🧭 26. «Jeg trenger protokolloppdagbarhet i produktets UX» +
+🔐 21. "I need scoped MCP permissions per integration" -Hvis brukere ikke kan oppdage protokolloverflater, faller kvaliteten på adopsjon og støtte. +Different clients should have least-privilege access to tool categories. -**Hvordan OmniRoute løser det:** +**How OmniRoute solves it:** -- Konsolidert**Endepunkter**-side med faner for proxy-, MCP-, A2A- og API-endepunkter -- Inline tjenestestatus veksler (Online/Offline) for MCP og A2A -- Lenker fra oversikt til dedikerte administrasjonsfaner
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling - -🧪 27. "Jeg trenger ende-til-ende protokollvalidering med ekte klienter" + -Mock-tester er ikke nok til å validere protokollkompatibilitet før utgivelse. +
+⚙️ 22. "I need operational controls without redeploying" -**Hvordan OmniRoute løser det:** +Teams need quick runtime changes during incidents or cost events. -- E2E-suite som starter opp app og bruker ekte MCP SDK-klienttransport -- A2A-klient tester for å oppdage, sende, streame, hente og kansellere flyter -- Krysssjekk påstander mot MCP-revisjon og A2A-oppgave-APIer
+**How OmniRoute solves it:** - -📡 28. «Jeg trenger enhetlig observerbarhet på tvers av alle grensesnitt» +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -Å dele observerbarhet etter protokoll skaper blinde flekker og lengre MTTR. + -**Hvordan OmniRoute løser det:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Samlede dashboards/logger/analyse i ett produkt -- Helse + revisjon + forespørsel om telemetri på tvers av OpenAI-, MCP- og A2A-lag -- Operasjonelle APIer for status og automatisering
+Without lifecycle visibility, task incidents become hard to triage. - -💼 29. "Jeg trenger én kjøretid for proxy + verktøy + agentorkestrering" +**How OmniRoute solves it:** -Å kjøre mange separate tjenester øker driftskostnadene og feilmodusene. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**Hvordan OmniRoute løser det:** + -- OpenAI-kompatibel proxy, MCP-server og A2A-server i én stabel -- Delt autentisering, robusthet, datalagring og observerbarhet -- Konsekvent policymodell på tvers av alle interaksjonsflater +
+🌊 24. "I need active stream metrics for A2A load" - -🚀 30. "Jeg trenger å sende agentiske arbeidsflyter uten limkodespredning" +Streaming workflows require operational insight into concurrency and live connections. -Lag mister hastighet når de setter sammen flere ad-hoc-tjenester og skript. +**How OmniRoute solves it:** -**Hvordan OmniRoute løser det:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Enhetlig endepunktstrategi for kunder og agenter -- Innebygde brukergrensesnitt for protokolladministrasjon og røykvalideringsveier -- Produksjonsklare fundamenter (sikkerhet, logging, robusthet, backup)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Playbook A: Maksimer betalt abonnement + billig backup**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -607,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Playbook B: Nullkostnadskodestabel**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Playbook C: 24/7 alltid aktiv reservekjede**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -630,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Playbook D: Agentoperasjoner med MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Konfigurer AI-koding på minutter til**$0/måned**. Koble til disse gratis kontoene og bruk den innebygde**Free Stack**-kombinasjonen. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Trinn | Handling | Leverandører ulåst | -| ---- | ---------------------------------------------------------- | -------------------------------------------------------------------------- | -| 1 | Koble til**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**ubegrenset**| -| 2 | Koble til**Qoder**(Google OAuth) | kimi-k2-tenkning, qwen3-coder-plus, deepseek-r1... —**ubegrenset**| -| 3 | Koble til**Qwen**(enhetskode) | qwen3-coder-pluss, qwen3-coder-flash... —**ubegrenset**| -| 4 | Koble til**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180K/mnd gratis**| -| 5 | `/dashboard/combos` →**Free Stack ($0)**mal | Round-robin alle gratisleverandører automatisk | +| Step | Action | Providers Unlocked | +| ---- | -------------------------------------------------- | ------------------------------------------------------------------ | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Pek en hvilken som helst IDE/CLI til:**`http://localhost:20128/v1` · API-nøkkel: `any-string` · Ferdig. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Valgfri ekstra dekning (også gratis):**Groq API-nøkkel (30 RPM gratis), NVIDIA NIM (40 RPM gratis, 70+ modeller), Cerebras (1M tok/dag), LongCat API-nøkkel (50M tokens/dag!), Cloudflare Workers AI (10K Neurons/day, 50+ modeller).## Hurtigstart +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Hurtigstart ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **pnpm-brukere:**Kjør `pnpm approve-builds -g` etter installasjon for å aktivere native build-skript som kreves av `better-sqlite3` og `@swc/core`: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > > ```bash -> pnpm installer -g omniroute -> pnpm approve-builds -g # Velg alle pakker → godkjenn +> pnpm install -g omniroute +> pnpm approve-builds -g # Select all packages → approve > omniroute > ``` -Dashboard åpnes på `http://localhost:20128` og API-base-URL er `http://localhost:20128/v1`. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Kommando | Beskrivelse | -| ----------------------- | ---------------------------------------------------------- | -| `omniroute` | Start server (`PORT=20128`, API og dashbord på samme port) | -| `omniroute --port 3000` | Sett kanonisk/API-port til 3000 | -| `omniroute --mcp` | Start MCP-server (stdio-transport) | -| `omniroute --no-open` | Ikke åpne nettleseren automatisk | -| `omniroute --help` | Vis hjelp | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Valgfri delt port-modus:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -For de fleste distribusjoner trenger du bare: +For most deployments, you only need: -| Variabel | Standard | Formål | -| -------------------------- | ------------------------------ | ---------------------------------------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600 000` | Delt grunnlinje for oppstrøms henting, skjulte Undici-tidsavbrudd, TLS-fingeravtrykkforespørsler og API-broforespørsel/proxy-tidsavbrudd | -| `STREAM_IDLE_TIMEOUT_MS` | arver `REQUEST_TIMEOUT_MS` | Maksimalt gap mellom streaming-biter før OmniRoute avbryter SSE-strømmen | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -Bakoverkompatibilitet er bevart: eksisterende `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` og andre tidsavbrudd per lag fungerer fortsatt og overstyrer den delte grunnlinjen. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Avanserte overstyringer er tilgjengelige hvis du trenger bedre kontroll:| Variabel | Standard | Formål | -| ------------------------------------------ | ------------------------------------------ | ---------------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | arver `REQUEST_TIMEOUT_MS` | Total tidsavbrudd for oppstrømsforespørsel brukt av hovedhentingsavbruddssignalet | -| `FETCH_HEADERS_TIMEOUT_MS` | arver `FETCH_TIMEOUT_MS` | Undici tidsbegrensning for mottak av oppstrøms svarhoder | -| `FETCH_BODY_TIMEOUT_MS` | arver `FETCH_TIMEOUT_MS` | Undici tidsbegrensning mellom oppstrøms kroppsdeler (`0` deaktiverer det) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30 000` | Undici TCP tilkobling tidsavbrudd | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici inaktiv hold-alive socket timeout | -| `TLS_CLIENT_TIMEOUT_MS` | arver `FETCH_TIMEOUT_MS` | Tidsavbrudd for TLS-fingeravtrykksforespørsler gjort gjennom `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | arver `REQUEST_TIMEOUT_MS` eller `30000` | Tidsavbrudd for "/v1" proxy-videresending fra API-port til dashbordport | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `maks(API_BRIDGE_PROXY_TIMEOUT_MS, 300 000)` | Tidsavbrudd for innkommende forespørsel på API-broserveren | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60 000` | Tidsavbrudd for innkommende overskrift på API-broserveren | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive-tidsavbrudd på API-broserveren | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Tidsavbrudd for socketinaktivitet på API-broserveren (`0` deaktiverer den) | +Advanced overrides are available if you need finer control: -Hvis du kjører OmniRoute bak Nginx, Caddy, Cloudflare eller en annen omvendt proxy, sørg for at proxyen -tidsavbrudd er også høyere enn OmniRoute strøm-/hentingstidsavbrudd.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Åpne Dashboard → `Providers` og koble til minst én leverandør (OAuth- eller API-nøkkel). -2. Åpne Dashboard → `Endepunkter` og lag en API-nøkkel. -3. (Valgfritt) Åpne Dashboard → `Combos` og angi reservekjeden.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Fungerer med Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode og OpenAI-kompatible SDK-er.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (for verktøydrevne operasjoner):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -Koble deretter MCP-klienten din over "stdio" og testverktøy som: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (for agent-til-agent arbeidsflyter):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -759,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Denne suiten validerer ekte MCP- og A2A-klientflyter mot en kjørende app.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -767,13 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` - -Ugyldig Linux (`xbps-src`-mal) +
+Void Linux (`xbps-src` template) -For Void Linux-brukere kan du bygge en innebygd pakke ved å bruke `xbps-src`. Lagre denne blokken som `srcpkgs/omniroute/mal`:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -785,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -793,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -869,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -880,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute er tilgjengelig som et offentlig Docker-bilde på [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Rask løp:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -890,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**Med miljøfil:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Bruke Docker Compose:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Dashboard-støtte for Docker-implementeringer inkluderer nå en ett-klikks**Cloudflare Quick Tunnel**på `Dashboard → Endpoints`. Den første aktiveringen laster bare ned `cloudflared` når det er nødvendig, starter en midlertidig tunnel til ditt nåværende `/v1`-endepunkt, og viser den genererte `https://*.trycloudflare.com/v1`-URLen rett under din vanlige offentlige URL. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Merknader: +Notes: -- Hurtigtunnel-URL-er er midlertidige og endres etter hver omstart. -- Hurtigtunneler gjenopprettes ikke automatisk etter omstart av OmniRoute eller container. Aktiver dem på nytt fra dashbordet ved behov. -- Administrert installasjon støtter for øyeblikket Linux, macOS og Windows på `x64` / `arm64`. -- Managed Quick Tunnels standard til HTTP/2-transport for å unngå støyende QUIC UDP-buffervarsler i begrensede containermiljøer. Angi `CLOUDFLARED_PROTOCOL=quic` eller `auto` hvis du vil ha en annen transport. -- Docker-bilder samler systemets CA-røtter og sender dem til administrert `cloudflared`, noe som unngår TLS-tillitsfeil når tunnelen starter opp inne i containeren. -- SQLite kjører i WAL-modus. `docker stop` bør få fullføres slik at OmniRoute kan sjekke de siste endringene tilbake til `storage.sqlite`. -- De medfølgende Compose-filene har allerede satt en utsettelsesperiode på 40-tallet. Hvis du kjører bildet direkte, behold `--stop-timeout 40` (eller lignende) slik at manuelle stopp ikke avskjærer oppryddingen. -- Angi `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` hvis du vil at OmniRoute skal bruke en eksisterende binær i stedet for å laste ned en. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Bruke Docker Compose med Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute kan eksponeres sikkert ved hjelp av Caddys automatiske SSL-klargjøring. Sørg for at domenets DNS A-post peker til serverens IP.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` +| Image | Tag | Size | Description | +| ------------------------ | -------- | ------ | --------------------- | +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | -| Bilde | Tag | Størrelse | Beskrivelse | -| -------------------------- | -------- | ------ | ---------------------- | -| `diegosouzapw/omniroute` | `siste` | ~250MB | Siste stabile utgivelse | -| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Gjeldende versjon |--- +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**NYHET!**OmniRoute er nå tilgjengelig som en**native desktop-applikasjon**for Windows, macOS og Linux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Kjør OmniRoute som en frittstående skrivebordsapp – ingen terminal, ingen nettleser, ingen internett nødvendig for lokale modeller. Den elektronbaserte appen inkluderer: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Native Window**— Dedikert appvindu med systemstatusfeltintegrasjon -- 🔄**Autostart**— Start OmniRoute ved systempålogging -- 🔔**Native notifications**— Få varsler for kvotebruk eller leverandørproblemer -- ⚡**One-Click Install**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Frakoblet modus**— Fungerer helt offline med medfølgende server### Hurtigstart +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Hurtigstart ```bash # Development mode @@ -979,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -Når den er minimert, lever OmniRoute i systemstatusfeltet med raske handlinger: +When minimized, OmniRoute lives in your system tray with quick actions: -- Åpne dashbordet -- Endre serverport -- Avslutt programmet +- Open dashboard +- Change server port +- Quit application -📖 Full dokumentasjon: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Nivå | Leverandør | Kostnad | Kvote Tilbakestill | Best for | -| ----------------- | --------------------------- | --------------------------------- | ----------------------- | --------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -| **💳 ABONNEMENT** | Claude Code (Pro) | $20/md | 5t + ukentlig | Allerede abonnert | -| | Codex (Pluss/Pro) | $20-200/md | 5t + ukentlig | OpenAI-brukere | -| | Gemini CLI | **GRATIS** | 180K/mnd + 1K/dag | Alle sammen! | -| | GitHub Copilot | $10-19/md | Månedlig | GitHub-brukere | -| **🔑 API NØKKEL** | NVIDIA NIM | **GRATIS**(utvikler for alltid) | ~40 RPM | 70+ åpne modeller | -| | Cerebras | **GRATIS**(1M tok/dag) | 60K TPM / 30 RPM | Verdens raskeste | -| | Groq | **GRATIS**(30 RPM) | 14,4K RPD | Ultrarask Llama/Gemma | -| | DeepSeek V3.2 | $0,27/$1,10 per 1M | Ingen | Beste pris/kvalitet resonnement | -| | xAI Grok-4 Fast | **$0,20/$0,50 per 1M**🆕 | Ingen | Raskeste + verktøykall, ultralavt | -| | xAI Grok-4 (standard) | $0,20/$1,50 per 1M 🆕 | Ingen | Resonneringsflaggskip fra xAI | -| | Mistral | Gratis prøveversjon + betalt | Begrenset pris | Europeisk AI | -| | OpenRouter | Betal per bruk | Ingen | 100+ modeller aggr. | -| **💰 BILLIG** | GLM-5 (via Z.AI) 🆕 | $0,5/1M | Daglig 10:00 | 128K utgang, nyeste flaggskip | -| | GLM-4.7 | $0,6/1M | Daglig 10:00 | Budsjett backup | -| | MiniMax M2.5 🆕 | $0,3/1 million input | 5-timers rullende | Resonnement + agentoppgaver | -| | MiniMax M2.1 | $0,2/1M | 5-timers rullende | Billigste alternativ | -| | Kimi K2.5 (Moonshot API) 🆕 | Betal per bruk | Ingen | Direkte Moonshot API-tilgang | -| | Kimi K2 | $9/md leilighet | 10 millioner tokens/mnd | Forutsigbar kostnad | -| **🆓 GRATIS** | Qoder | **$0** | Ubegrenset | 5 modeller ubegrenset | -| | Qwen | **$0** | Ubegrenset | 4 modeller ubegrenset | -| | Kiro | **$0** | Ubegrenset | Claude Sonnet/Haiku (AWS-bygger) | -| | LongCat Flash-Lite 🆕 | **$0**(50 millioner tok/dag 🔥) | 1 RPS | Største gratis kvote på jorden | -| | Pollinasjoner AI 🆕 | **$0**(ingen nøkkel nødvendig) | 1 krav/15s | GPT-5, Claude, DeepSeek, Llama 4 | -| | Cloudflare Workers AI 🆕 | **$0**(10K nevroner/dag) | ~150 resp/dag | 50+ modeller, global kant | -| | Scaleway AI 🆕 | **$0**(1 millioner tokens totalt) | Begrenset pris | EU/GDPR, Qwen3 235B, Lama 70B | > 🆕**Nye modeller lagt til (mars 2026):**Grok-4 Fast-familie til $0,20/$0,50/M (benchmarked ved 1143ms — 30 % raskere enn Gemini 2.5 Flash), GLM-5 via Z.AI med 128K-utgang, MiniMax M2.5-resonnement, KimSeidek pr V3.2 update, KimSeidek pr. direkte API. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 $0 Combo Stack — Det komplette gratis oppsettet:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Null kostnad. Slutter aldri å kode.**Konfigurer dette som én OmniRoute-kombinasjon og alle fallbacks skjer automatisk – ingen manuell veksling noensinne.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Alle modellene nedenfor er**100 % gratis uten behov for kredittkort**. OmniRoute ruter automatisk mellom dem når én kvote går tom – kombiner dem alle for en ubrytelig kombinasjon av $0.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Modell | Prefiks | Grens | Satsgrense | -| ------------------ | ------ | ------------- | ---------------------- | -| `claude-sonnett-4.5` | `kr/` |**Ubegrenset**| Ingen rapportert daglig tak | -| `claude-haiku-4.5` | `kr/` |**Ubegrenset**| Ingen rapportert daglig tak | -| `claude-opus-4.6` | `kr/` |**Ubegrenset**| Siste Opus via Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) -| Modell | Prefiks | Grens | Satsgrense | +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | --------------------- | +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | + +### 🟢 QODER MODELS (Free PAT via qodercli) + +| Model | Prefix | Limit | Rate Limit | | ------------------ | ------ | ------------- | --------------- | -| `kimi-k2-tenking` | `hvis/` |**Ubegrenset**| Ingen rapportert cap | -| `qwen3-coder-pluss` | `hvis/` |**Ubegrenset**| Ingen rapportert tak | -| `deepseek-r1` | `hvis/` |**Ubegrenset**| Ingen rapportert cap | -| `minimax-m2.1` | `hvis/` |**Ubegrenset**| Ingen rapportert tak | -| `kimi-k2` | `hvis/` |**Ubegrenset**| Ingen rapportert tak | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -> Anbefalt tilkoblingsmetode:**Personal Access Token + `qodercli`**. Nettleser OAuth er -> eksperimentell og deaktivert som standard med mindre `QODER_OAUTH_*` miljøvariabler er konfigurert.### 🟡 QWEN MODELS (Device Code Auth) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| Modell | Prefiks | Grens | Satsgrense | -| ------------------ | ------ | ------------- | ------------------ | -| `qwen3-coder-pluss` | `qw/` |**Ubegrenset**| Ingen rapportert cap | -| `qwen3-coder-flash` | `qw/` |**Ubegrenset**| Ingen rapportert cap | -| `qwen3-coder-next` | `qw/` |**Ubegrenset**| Ingen rapportert tak | -| `vision-model` | `qw/` |**Ubegrenset**| Multimodal (bilder) |### 🟣 GEMINI CLI (Google OAuth) +### 🟡 QWEN MODELS (Device Code Auth) -| Modell | Prefiks | Begrensning | Satsgrense | -| -------------------------- | ------ | -------------------------- | ------------- | -| `gemini-3-flash-preview` | `gc/` |**180K tok/måned**+ 1K/dag | Månedlig tilbakestilling | -| `gemini-2.5-pro` | `gc/` | 180K/måned (delt basseng) | Høy kvalitet |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | ------------------- | +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Nivå | Daglig grense | Satsgrense | Merknader | -| ---------- | ------------ | ----------- | -------------------------------------------------------------- | -| Gratis (Dev) | Ingen token cap |**~40 RPM**| 70+ modeller; overgang til rene takstgrenser medio 2025 | +### 🟣 GEMINI CLI (Google OAuth) -Populære gratismodeller: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-seek-instruct`/`deepseek-seek`/`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -| Nivå | Daglig grense | Satsgrense | Merknader | -| ---- | ------------------ | ---------------- | -------------------------------------------------- | -| Gratis |**1 millioner tokens/dag**| 60K TPM / 30 RPM | Verdens raskeste LLM-slutning; tilbakestilles daglig | +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) -Gratis tilgjengelig: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-destill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +| Tier | Daily Limit | Rate Limit | Notes | +| ---------- | ------------ | ----------- | ------------------------------------------------------ | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -| Nivå | Daglig grense | Satsgrense | Merknader | -| ---- | ------------- | ---------------- | ------------------------------------------ | -| Gratis |**14,4K RPD**| 30 RPM per modell | Ingen kredittkort; 429 på grense, ikke belastet | +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -Gratis tilgjengelig: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -| Modell | Prefiks | Daglig gratis kvote | Merknader | -| ------------------------------ | ------ | ------------------ | ---------------------------- | -| `LongCat-Flash-Lite` | `lc/` |**50 millioner tokens**💥 | Største gratis kvote noensinne | -| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | -| `LongCat-Flash-tenker` | `lc/` | 500K tokens | Begrunnelse / CoT | -| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 versjon | -| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -> 100 % gratis mens du er i offentlig beta. Registrer deg på [longcat.chat](https://longcat.chat) med e-post eller telefon. Tilbakestilles daglig 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -| Modell | Prefiks | Satsgrense | Leverandøren bak | +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ------------- | ---------------- | ----------------------------------------- | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | + +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` + +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 + +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | + +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. + +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| `openai` | `pol/` | 1 krav/15s | GPT-5 | -| `claude` | `pol/` | 1 krav/15s | Antropiske Claude | -| `tvilling` | `pol/` | 1 krav/15s | Google Gemini | -| `deepseek` | `pol/` | 1 krav/15s | DeepSeek V3 | -| `llama` | `pol/` | 1 krav/15s | Meta Llama 4 speider | -| `mistral` | `pol/` | 1 krav/15s | Mistral AI | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**Null friksjon:**Ingen registrering, ingen API-nøkkel. Legg til pollineringsleverandøren med et tomt nøkkelfelt og det fungerer umiddelbart.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| Nivå | Daglige nevroner | Tilsvarende bruk | Merknader | -| ---- | ------------- | ----------------------------------------------- | ---------------------------- | -| Gratis |**10 000**| ~150 LLM resp / 500s lyd / 15K innebygging | Global edge, 50+ modeller | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 -Populære gratismodeller: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (gratis lyd!), `@cf/qwen/qwen2.5-coder-15b-instruct` +| Tier | Daily Neurons | Equivalent Usage | Notes | +| ---- | ------------- | --------------------------------------- | ----------------------- | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -> Krever API-token + konto-ID fra [dash.cloudflare.com](https://dash.cloudflare.com). Lagre konto-ID i leverandørinnstillingene.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -| Nivå | Gratis kvote | Plassering | Merknader | -| ---- | ------------- | ------------ | ---------------------------------- | -| Gratis |**1 millioner tokens**| 🇫🇷 Paris, EU | Ingen kredittkort nødvendig innenfor grensene | +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -Gratis tilgjengelig: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 -> EU/GDPR-kompatibel. Få API-nøkkel på [console.scaleway.com](https://console.scaleway.com). +| Tier | Free Quota | Location | Notes | +| ---- | ------------- | ------------ | ----------------------------------- | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | ->**💡 Den ultimate gratisstakken (11 leverandører, $0 for alltid):** +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` + +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). + +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Sonnet/Haiku UBEGRENSET -> Qoder (hvis/) → kimi-k2-tenkning, qwen3-koder-pluss, deepseek-r1 UBEGRENSET -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 millioner tokens/dag 🔥 -> Pollinasjoner (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — ingen nøkkel nødvendig -> Qwen (qw/) → qwen3-kodermodeller UBEGRENSET -> Gemini (gemini/) → Gemini 2.5 Flash — 1500 rekv/dag gratis -> Cloudflare AI (jf/) → 50+ modeller — 10K nevroner/dag -> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M gratis tokens (EU) -> Groq (groq/) → Lama/Gemma — 14,4K req/dag ultrarask -> NVIDIA NIM (nvidia/) → 70+ åpne modeller — 40 RPM for alltid -> Cerebras (cerebras/) → Lama/Qwen verdens raskeste — 1M tok/dag -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Transkriber hvilken som helst lyd/video for**$0**— Deepgram-emner med $200 gratis, AssemblyAI $50 reserve, Groq Whisper som ubegrenset nødbackup. +## 🎙️ Free Transcription Combo -| Leverandør | Gratis kreditter | Beste modell | Satsgrense | -| ------------------ | ---------------------------- | -------------------------------------------- | ---------------------------- | -|**Deepgram**|**$200 gratis**(påmelding) | `nova-3` — beste nøyaktighet, 30+ språk | Ingen RPM-grense på gratis kreditter | -| 🔵**AssemblyAI**|**$50 gratis**(påmelding) | `universal-3-pro` — kapitler, sentiment, PII | Ingen RPM-grense på gratis kreditter | -| 🔴**Groq**|**Gratis for alltid**| `whisper-large-v3` — OpenAI Whisper | 30 RPM (hastighetsbegrenset) | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. -**Foreslått kombinasjon i `/dashboard/combos`:**``` +| Provider | Free Credits | Best Model | Rate Limit | +| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | + +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Deretter i `/dashboard/media` →**Transkripsjon**-fanen: last opp hvilken som helst lyd- eller videofil → velg det kombinasjonsendepunktet → få transkripsjon i støttede formater.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 er bygget som en operativ plattform, ikke bare en reléproxy.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Funksjon | Hva det gjør | -| --------------------------------------- | ---------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Grok-4 Fast Family** | xAI-modeller til $0,20/$0,50/M — benchmarked 1143ms (30 % raskere enn Gemini 2.5 Flash) | -| 🧠**GLM-5 via Z.AI** | 128K utdatakontekst, $0,5/1M — nyeste flaggskip fra GLM-familien | -| 🔮**MiniMax M2.5** | Resonnering + agentoppgaver til $0,30/1M — betydelig oppgradering fra M2.1 | -| 🎯**verktøy Calling Flag per modell** | Per-modell `toolCalling: true/false` i registeret — AutoCombo hopper over ikke-verktøy-kompatible modeller | -| 🌍**Flerspråklig hensiktsgjenkjenning** | PT/ZH/ES/AR nøkkelord i AutoCombo scoring — bedre modellvalg for ikke-engelsk innhold | -| 📊**Referansedrevne fallbacks** | Ekte p95-forsinkelse fra live-forespørsler feeds combo scoring — AutoCombo lærer av faktiske data | -| 🔁**Be om deduplisering** | Innhold-hash-basert dedup-vindu – multi-agent trygt, forhindrer dupliserte belastninger | -| 🔌**Plugbar ruterstrategi** | Utvidbart `RouterStrategy`-grensesnitt – legg til tilpasset rutinglogikk som plugins | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Funksjon | Hva det gjør | -| -------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Modell lekeplass** | Dashboard-side for å teste hvilken som helst modell direkte — leverandør/modell/endepunktvelgere, Monaco Editor, streaming, avbryt, timing | -| 🔏**CLI Fingerprint Matching** | Bestilling av topptekst/kropp per leverandør for å matche native CLI-signaturer – bytt per leverandør i Innstillinger > Sikkerhet.**Din proxy-IP er bevart** | -| 🤝**ACP Support (Agent Client Protocol)** | CLI-agentoppdagelse (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 flere), prosessoppstarter, `/api/acp/agents` endepunkt | -| 🤖**ACP Agents Dashboard** | Feilsøking › Agenter-side — rutenett med 14 agenter med installasjonsstatus, versjon, tilpasset agentskjema for ethvert CLI-verktøy.**OpenCode**-brukere får en "Last ned opencode.json"-knapp som automatisk genererer en klar-til-bruk konfig med alle tilgjengelige modeller. | -| 🔧**Egendefinert modell `apiFormat`-ruting** | Egendefinerte modeller med `apiFormat: "responses"` rutes nå riktig til Responses API-oversetteren | -| 🏢**Codex Workspace Isolering** | Flere Codex-arbeidsområder per e-post — OAuth skiller tilkoblinger riktig etter arbeidsområde-ID | -| 🔄**Automatisk oppdatering av elektroner** | Desktop-app ser etter oppdateringer + automatisk installering ved omstart | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Funksjon | Hva det gjør | -| --------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**MCP-server (25 verktøy)** | IDE/agent-verktøy via 3 transporter: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 kjerner + 3 minne + 4 ferdighetsverktøy | -| 🤝**A2A-server (JSON-RPC + SSE)** | Agent-til-agent oppgavekjøring med synkronisering og strømmeflyter | -| 🧭**Side for konsoliderte endepunkter** | Fanebasert administrasjonsside med Endpoint Proxy, MCP, A2A og API Endpoints faner | -| 🎚️**Tjenesteaktiver/deaktiver brytere** | PÅ/AV-brytere for MCP og A2A med innstillingsfasthet (standard: AV) | -| 🛰️**MCP Runtime Heartbeat** | Reell prosessstatus (pid, oppetid, hjerteslagsalder, transport, omfangsmodus) | -| 📋**MCP Audit Trail** | Filtrerbare revisjonslogger med suksess/fiasko og nøkkelattribusjon | -| 🔐**MCP Scope Enforcement** | 10 granulære omfangstillatelser for kontrollert verktøytilgang | -| 📡**A2A Task Lifecycle Management** | List/filtrer oppgaver, inspiser hendelser/artefakter, avbryt kjørende oppgaver | -| 📋**Agent Card Discovery** | `/.well-known/agent.json` for automatisk oppdagelse av klient | -| 🧪**Protocol E2E Test Sele** | Ekte MCP SDK + A2A-klient flyter i `test:protocols:e2e` | -| ⚙️**Operasjonskontroller** | Bytt kombinasjon, påfør elastisitetsprofiler, tilbakestill brytere fra én kontrollflate | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Funksjon | Hva det gjør | -| ---------------------------------- | --------------------------------------------------------------------------- | ----------------------- | -| 🎯**Smart 4-lags fallback** | Automatisk rute: Abonnement → API-nøkkel → Billig → Gratis | -| 📊**Sanntidskvotesporing** | Live tokenantall + tilbakestilt nedtelling per leverandør | -| 🔄**Formatoversettelse** | OpenAI ↔ Claude ↔ Gemini ↔ Svar med skjemasikre konverteringer | -| 👥**Støtte for flere kontoer** | Flere kontoer per leverandør med intelligent utvalg | -| 🔄**Auto Token Refresh** | OAuth-tokens oppdateres automatisk med prøv på nytt | -| 🎨**Egendefinerte kombinasjoner** | 9 balansestrategier + reservekjedekontroll | -| 🌐**Wildcard-ruter** | `leverandør/*` dynamisk ruting | -| 🧠**Tenker budsjettkontroller** | Begrensninger for gjennomgang, automatisk, tilpasset og adaptiv resonnement | -| 🔀**Modellaliaser** | Innebygd + tilpasset modellaliasing og migrasjonssikkerhet | -| ⚡**Bakgrunnsforringelse** | Rut lavprioriterte bakgrunnsoppgaver til billigere modeller | -| 🧪**Task-Aware Smart Ruting** | Auto-velg modell etter innholdstype (koding/visjon/analyse/oppsummering) | -| 🔄**A2A-agentarbeidsflyt** | Deterministisk FSM-orkestrator for stateful multi-step agent henrettelser | -| 🔀**Adaptiv ruting** | Dynamisk strategioverstyring basert på tokenvolum og promptkompleksitet | -| 🎲**Tilbydermangfold** | Shannon entropi scoring balansering av auto-combo trafikkdistribusjon | -| 💬**Systemprompt-injeksjon** | Globale atferdskontroller brukes konsekvent | -| 📄**Responses API-kompatibilitet** | Full `/v1/responses`-støtte for Codex og avanserte agentarbeidsflyter | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Funksjon | Hva det gjør | -| ---------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ---------------------------------------- | -| 🖼️**Bildegenerering** | `/v1/images/generations` med sky og lokale backends | -| 📐**Innbygging** | `/v1/embeddings` for søk og RAG-rørledninger | -| 🎤**Lydtranskripsjon** | `/v1/audio/transcriptions` — 7 leverandører (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), automatisk språkdeteksjon, MP4/MP3/WAV-støtte | -| 🔊**Tekst-til-tale** | `/v1/audio/speech` — 10 leverandører (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) med riktige feilmeldinger | -| 🎬**Videogenerering** | `/v1/videos/generations` (ComfyUI + SD WebUI arbeidsflyter) | -| 🎵**Musikkgenerasjon** | `/v1/music/generations` (ComfyUI-arbeidsflyter) | -| 🛡️**Moderasjoner** | `/v1/moderations` sikkerhetssjekker | -| 🔀**Omrangering** | `/v1/rerank` for relevansscoring | -| 🔍**Nettsøk**🆕 | `/v1/search` — 5 leverandører (Serper, Brave, Perplexity, Exa, Tavily), 6500+ gratis/måned, auto-failover, cache | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Funksjon | Hva det gjør | -| --------------------------------------------- | -------------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**Maksimalbrytere** | Per modell tur/gjenoppretting med terskelkontroller | -| 🎯**Endepunktbevisste modeller** | Egendefinerte modeller erklærer støttede endepunkter + API-format | -| 🛡️**Anti-tordenflokk** | Mutex + semaforbeskyttelse ved forsøk på nytt/rate hendelser | -| 🧠**Semantisk + signaturbuffer** | Kostnads-/forsinkelsesreduksjon med to hurtigbufferlag | -| ⚡**Be om idempotens** | Duplisert beskyttelsesvindu | -| 🔒**TLS-fingeravtrykkspoofing** | Nettleserlignende TLS-fingeravtrykk —**reduserer botdeteksjon og kontoflagging** | -| 🔏**CLI Fingerprint Matching** | Matcher native CLI-forespørselssignaturer —**reduserer utestengelsesrisiko mens proxy-IP bevares** | -| 🌐**IP-filtrering** | Tillatelsesliste/blokkeringslistekontroll for eksponerte distribusjoner | -| 📊**Redigerbare satsgrenser** | Konfigurerbare grenser på globalt nivå/leverandørnivå med utholdenhet | -| 📉**Grasiøs nedbrytning** | Fallbacks med flerlagskapasitet som beskytter kjernegatewayoperasjoner | -| 📜**Config Audit Trail** | Diff-basert endringssporing forhindrer driftsdrift med enkle tilbakeføringer | -| ⏳**Provider Health Sync** | Proaktiv token-utløpsovervåking som utløser varsler før autorisasjonsfeil | -| 🚪**Automatisk deaktiver utestengte kontoer** | Driftsbryter forsegling permanent blokkerte token-kontoer automatisk | -| 🔑**API Key Management + Scoping** | Sikker nøkkelutstedelse/rotasjon og modell/leverandørkontroller | -| 👁️**Scoped API Key Reveal**🆕 | Opt-in gjenoppretting av API-nøkler via `ALLOW_API_KEY_REVEAL` | -| 🛡️**Beskyttet `/modeller`** | Valgfri godkjenning og leverandørskjul for modellkatalog | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Funksjon | Hva det gjør | -| --------------------------------- | ------------------------------------------------------------------ | ---------------------------- | -| 📝**Forespørsel + proxy-logging** | Full forespørsel/svar og proxy-logging | -| 📉**Strømde detaljerte logger**🆕 | Rekonstruerer SSE-nyttelaststrømmer rent inn i brukergrensesnittet | -| 📋**Unified Logs Dashboard** | Forespørsels-, proxy-, revisjons- og konsollvisninger på én side | -| 🔍**Be om telemetri** | p50/p95/p99 ventetid og forespørselssporing | -| 🏥**Helse Dashboard** | Oppetid, breaker-tilstander, lockouts, cachestatistikk | -| 💰**Kostnadssporing** | Budsjettkontroller og prissetting per modell | -| 📈**Analytics-visualiseringer** | Modell-/leverandørbruksinnsikt og trendvisninger | -| 🧪**Evalueringsramme** | Gylden sett-testing med konfigurerbare kampstrategier | -| 📡**Live Diagnostics**🆕 | Semantisk cache-bypass for nøyaktig combo live testing | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Funksjon | Hva det gjør | -| ---------------------------------------- | ------------------------------------------------------------------ | --------------------- | -| 🌐**Distribuer hvor som helst** | Localhost, VPS, Docker, Cloud-miljøer | -| 🚇**Cloudflare Tunnel**🆕 | Ett-klikks Quick Tunnel-integrasjon fra dashbordet | -| 🔑**API-nøkkelmodellfiltrering** | Native /v1/modeller-svar filtrert via tildelte bærerkontekstroller | -| ⚡**Smart Cache Bypass** | Konfigurerbar TTL-heuristikk og tvungen gjenhentingskontroller | -| 🔄**Sikkerhetskopiering/gjenoppretting** | Eksport/import og gjenopprettingsflyter | -| 🧙**Onboarding Wizard** | Førstegangs veiledet oppsett | -| 🔧**CLI Tools Dashboard** | Ett-klikks oppsett for populære kodeverktøy | -| 🎮**Modell lekeplass** | Test hvilken som helst leverandør/modell/endepunkt fra dashbordet | -| 🔏**CLI Fingerprint Toggle** | Fingeravtrykksmatching per leverandør i Innstillinger > Sikkerhet | -| 🌐**i18n (30 språk)** | Fullt dashbord + støtte for dokumentspråk med RTL-dekning | -| 🧹**Slett alle modeller** | Sletting av modellliste med ett klikk i leverandørdetaljer | -| 👁️**Sidefeltkontroller**🆕 | Skjul komponenter og integrasjoner fra Utseendeinnstillinger | -| 📋**Utgavemaler** | Standardiserte GitHub-maler for feil og funksjoner | -| 📂**Tilpasset datakatalog** | `DATA_DIR` overstyring for lagringssted | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1292,103 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -Når kvote, sats eller helse svikter, flytter OmniRoute automatisk til neste kandidat uten manuelt bytte.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A kan oppdages i brukergrensesnitt og dokumenter (ikke skjult) -- Protokollstatus-APIer avslører live driftsdata (`/api/mcp/*`, `/api/a2a/*`) -- Dashboards inkluderer handlinger for dag-2 operasjoner (kombinasjonsveksler, tilbakestilling av brytere, kansellering av oppgaver)#### Translator + validation workflow +#### Protocol management that is visible and operable -Oversetterområdet inkluderer: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Lekeplass**: be om transformasjonssjekker -**Chattetester**: full forespørsel/svar tur/retur -**Testbenk**: flere kofferter i en kjøring -**Live Monitor**: trafikkvisning i sanntid +#### Translator + validation workflow -Pluss protokollvalidering med ekte klienter via `npm run test:protocols:e2e`. +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Verktøyreferanse, IDE-konfigurasjoner og klienteksempler +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A Server README](src/lib/a2a/README.md)**— Ferdigheter, JSON-RPC-metoder, strømming og oppgavelivssyklus## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute inkluderer et innebygd evalueringsrammeverk for å teste LLM-responskvaliteten mot et gyldent sett. Få tilgang til den via**Analytics → Evals**i dashbordet.### Built-in Golden Set +## 🧪 Evaluations (Evals) -Det forhåndsinnlastede "OmniRoute Golden Set" inneholder testtilfeller for: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Hilsen, matematikk, geografi, kodegenerering -- Samsvar med JSON-format, oversettelse, generering av markdown -- Sikkerhetsavslag (skadelig innhold), telling, boolsk logikk### Evaluation Strategies +### Built-in Golden Set -| Strategi | Beskrivelse | Eksempel | -| -------------- | --------------------------------------------------------------------- | -------------------------------- | --- | -| `nøyaktig` | Utdata må samsvare nøyaktig med | `"4"` | -| `inneholder` | Utdata må inneholde understreng (uavhengig av store og små bokstaver) | `"Paris"` | -| `regex` | Utdata må samsvare med regulært uttrykksmønster | `"1.*2.*3"` | -| `egendefinert` | Egendefinert JS-funksjon returnerer true/false | `(output) => output.length > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) - -🧩 MCP-oppsett (Model Context Protocol) +
+🧩 MCP Setup (Model Context Protocol) -Start MCP-transport i stdio-modus:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Anbefalt valideringsflyt: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Koble til MCP-klienten over stdio. -2. Kjør `omniroute_get_health`. -3. Kjør `omniroute_list_combos`. -4. Åpne `/dashboard/mcp` for å bekrefte hjerteslag, aktivitet og revisjon. - -Nyttige APIer for automatisering: +Useful APIs for automation: - `GET /api/mcp/status` - `GET /api/mcp/tools` -- `GET /api/mcp/revisjon` -- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit` +- `GET /api/mcp/audit/stats` - -🤝 A2A-oppsett (Agent2Agent) + -Oppdag agenten:```bash +
+🤝 A2A Setup (Agent2Agent) + +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Send en oppgave:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` - -Administrer livssyklus: +Manage lifecycle: - `GET /api/a2a/status` -- `GET /api/a2a/oppgaver` +- `GET /api/a2a/tasks` - `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -Driftsgrensesnitt: +Operational UI: -- `/dashboard/a2a` for observerbarhet for oppgave/tilstand/strøm og røykhandlinger
+- `/dashboard/a2a` for task/state/stream observability and smoke actions - -🧪 End-to-end protokollvalidering + -Valider begge protokollene med ekte klienter:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Dette bekrefter: +This verifies: -- MCP SDK-klient koble/liste/ringe -- A2A-oppdagelse/send/stream/hent/avbryt -- Krysssjekk data i MCP-revisjon og A2A-oppgavestyrings-APIer
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs - -💳 Abonnementsleverandører### Claude Code (Pro/Max) + + +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1401,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Profftips:**Bruk Opus for komplekse oppgaver, Sonnet for hastighet. OmniRoute sporer kvote per modell!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1415,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Hver Codex-konto har nå policy-veksler i `Dashboard -> Providers`: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (PÅ/AV): håndhev 5-timers vindusterskelpolicyen. -- "Ukentlig" (PÅ/AV): håndhev terskelpolicyen for ukentlige vindu. -- Terskeladferd: når et aktivert vindu når >=90 % bruk, hoppes den kontoen over. -- Rotasjonsatferd: OmniRoute ruter automatisk til neste kvalifiserte Codex-konto. -- Tilbakestill atferd: når leverandøren 'resetAt'-tiden går, blir kontoen automatisk kvalifisert igjen. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Scenarier: +Scenarios: -- `5h PÅ` + `Ukentlig PÅ`: kontoen hoppes over når et av vinduene når terskelen. -- `5h OFF` + `Weekly ON`: bare ukentlig bruk kan blokkere kontoen. -- `5h PÅ` + `Ukentlig AV`: bare 5-timers bruk kan blokkere kontoen. -- `resetAt` bestått: kontoen går automatisk inn i rotasjonen på nytt (ingen manuell reaktivering).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1440,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Mest verdi:**Enormt gratis nivå! Bruk dette før betalte nivåer.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1455,71 +1662,91 @@ Models:
- -🔑 API nøkkelleverandører### NVIDIA NIM (FREE developer access — 70+ models) +
+🔑 API Key Providers -1. Registrer deg: [build.nvidia.com](https://build.nvidia.com) -2. Få gratis API-nøkkel (1000 slutningspoeng inkludert) -3. Dashboard → Legg til leverandør → NVIDIA NIM: - - API-nøkkel: `nvapi-din-nøkkel` +### NVIDIA NIM (FREE developer access — 70+ models) -**Modeller:**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` og 50+ flere +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Profftips:**OpenAI-kompatibel API — fungerer sømløst med OmniRoutes formatoversettelse!### DeepSeek +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -1. Registrer deg: [platform.deepseek.com](https://platform.deepseek.com) -2. Få API-nøkkel -3. Dashboard → Legg til leverandør → DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -**Modeller:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +### DeepSeek -1. Registrer deg: [console.groq.com](https://console.groq.com) -2. Få API-nøkkel (gratis nivå inkludert) -3. Dashboard → Legg til leverandør → Groq +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -**Modeller:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Profftips:**Ultrarask slutning — best for sanntidskoding!### OpenRouter (100+ Models) +### Groq (Free Tier Available!) -1. Registrer deg: [openrouter.ai](https://openrouter.ai) -2. Få API-nøkkel -3. Dashboard → Legg til leverandør → OpenRouter +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -**Modeller:**Få tilgang til 100+ modeller fra alle store leverandører gjennom én enkelt API-nøkkel. +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Dashboard-atferd:**OpenRouter-modeller administreres fra**Tilgjengelige modeller**. Manuell legg til, import og automatisk synkronisering oppdaterer alle den samme listen.
+**Pro Tip:** Ultra-fast inference — best for real-time coding! - -💰 Billige leverandører (backup)### GLM-4.7 (Daily reset, $0.6/1M) +### OpenRouter (100+ Models) -1. Registrer deg: [Zhipu AI](https://open.bigmodel.cn/) -2. Få API-nøkkel fra Coding Plan -3. Dashboard → Legg til API-nøkkel: - - Leverandør: `glm` - - API-nøkkel: `din-nøkkel` +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -**Bruk:**`glm/glm-4.7` +**Models:** Access 100+ models from all major providers through a single API key. -**Profftips:**Coding Plan tilbyr 3× kvote til 1/7 kostnad! Tilbakestill daglig 10:00.### MiniMax M2.1 (5h reset, $0.20/1M) +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -1. Registrer deg: [MiniMax](https://www.minimax.io/) -2. Få API-nøkkel -3. Dashboard → Legg til API-nøkkel + -**Bruk:**`minimax/MiniMax-M2.1` +
+💰 Cheap Providers (Backup) -**Profftips:**Billigste alternativet for lang kontekst (1M tokens)!### Kimi K2 ($9/month flat) +### GLM-4.7 (Daily reset, $0.6/1M) -1. Abonner: [Moonshot AI](https://platform.moonshot.ai/) -2. Få API-nøkkel -3. Dashboard → Legg til API-nøkkel +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**Bruk:**`kimi/kimi-latest` +**Use:** `glm/glm-4.7` -**Profftips:**Fast $9/måned for 10M tokens = $0,90/1M effektiv kostnad!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. - -🆓 GRATIS Leverandører (Emergency Backup)### Qoder (5 FREE models via OAuth) +### MiniMax M2.1 (5h reset, $0.20/1M) + +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `minimax/MiniMax-M2.1` + +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1560,8 +1787,10 @@ Models:
- -🎨 Lag kombinasjoner### Example 1: Maximize Subscription → Cheap Backup +
+🎨 Create Combos + +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1589,8 +1818,10 @@ Cost: $0 forever!
- -🔧 CLI-integrasjon### Cursor IDE +
+🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1601,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Bruk**CLI Tools**-siden i dashbordet for ett-klikks konfigurasjon, eller rediger `~/.claude/settings.json` manuelt.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1612,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Alternativ 1 – Dashboard (anbefalt):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Alternativ 2 — Manuell:**Rediger `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1629,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Merk:**OpenClaw fungerer bare med lokale OmniRoute. Bruk `127.0.0.1` i stedet for `localhost` for å unngå problemer med IPv6-oppløsning.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1643,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Trinn 1:**Legg til OmniRoute som en tilpasset leverandør:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Trinn 2:**Opprett/rediger `opencode.json` i prosjektroten din:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1669,117 +1909,130 @@ opencode } } } -```` +``` -**Trinn 3:**Velg modellen i OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Tips:**Legg til en hvilken som helst modell som er tilgjengelig i OmniRoute `/v1/models`-endepunktet til `modeller`-delen. Bruk formatet `provider/model-id` fra OmniRoute-dashbordet.
+ --- ## Feilsøking - -Klikk for å utvide feilsøkingsveiledningen +
+Click to expand troubleshooting guide -**«Språkmodellen ga ikke meldinger»** +**"Language model did not provide messages"** -- Leverandørkvoten er oppbrukt → Sjekk dashboardkvotesporing -- Løsning: Bruk kombinasjonsalternativ eller bytt til et billigere nivå +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**Satsbegrensning** +**Rate limiting** -- Abonnementskvote ut → Fallback til GLM/MiniMax -- Legg til kombinasjon: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**OAuth-token er utløpt** +**OAuth token expired** -- Automatisk oppdatering av OmniRoute -- Hvis problemene vedvarer: Dashboard → Leverandør → Koble til på nytt +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Høye kostnader** +**High costs** -- Sjekk bruksstatistikk i Dashboard → Kostnader -- Bytt primærmodell til GLM/MiniMax -- Bruk gratis nivå (Gemini CLI, Qoder) for ikke-kritiske oppgaver +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Dashboard/API-porter er feil** +**Dashboard/API ports are wrong** -- `PORT` er den kanoniske basisporten (og API-porten som standard) -- `API_PORT` overstyrer bare OpenAI-kompatibel API-lytter -- `DASHBOARD_PORT` overstyrer kun dashboard/Next.js-lytter -– Angi `NEXT_PUBLIC_BASE_URL` til dashbordet/den offentlige nettadressen din (for OAuth-tilbakeringing) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Skysynkroniseringsfeil** +**Cloud sync errors** -- Bekreft at "BASE_URL" peker på den kjørende forekomsten -- Bekreft `CLOUD_URL` peker til det forventede skyendepunktet -- Hold `NEXT_PUBLIC_*`-verdier på linje med verdiene på tjenersiden +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Første pålogging fungerer ikke** +**First login not working** -- Sjekk `INITIAL_PASSWORD` i `.env` -- Hvis det ikke er angitt, er reservepassordet "123456". +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Ingen forespørselslogger** +**No request logs** -- Forespørselsartefakter skrives til `DATA_DIR/call_logs/` som én JSON-fil per forespørsel -- Aktiver pipeline-fangst fra Dashboard → Logger → Forespørselslogger hvis du trenger detaljerte nyttelaster per trinn -- Angi `APP_LOG_TO_FILE=true` hvis du også vil ha applikasjonskonsolllogger i `logs/application/app.log` -– Juster «APP_LOG_MAX_FILE_SIZE», «APP_LOG_RETENTION_DAYS», «APP_LOG_MAX_FILES» og «CALL_LOG_MAX_ENTRIES» etter behov +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Test viser «Ugyldig» for OpenAI-kompatible leverandører** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Mange leverandører avslører ikke et `/modeller`-endepunkt -- OmniRoute v1.0.6+ inkluderer reservevalidering via chatfullføringer -- Sørg for at basis-URL inkluderer `/v1`-suffiks### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server - + ->**⚠️ Viktig for brukere som kjører OmniRoute på en VPS, Docker eller hvilken som helst ekstern server**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -**Antigravity**og**Gemini CLI**-leverandørene bruker**Google OAuth 2.0**. Google krever at «redirect_uri» i OAuth-flyten nøyaktig samsvarer med en av de forhåndsregistrerte URIene i appens Google Cloud Console. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -OAuth-legitimasjonen som er samlet i OmniRoute er registrert**kun for `localhost`**. Når du får tilgang til OmniRoute på en ekstern server (f.eks. `https://omniroute.myserver.com`), avviser Google autentiseringen med:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Du må opprette en**OAuth 2.0 Client ID**i Google Cloud Console med serverens URI.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Åpne Google Cloud Console** +#### Step-by-step -Gå til: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Opprett en ny OAuth 2.0-klient-ID** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Klikk**"+ Opprett legitimasjon"**→**"OAuth-klient-ID"** -- Søknadstype:**"Nettapplikasjon"** -- Navn: alt du liker (f.eks. "OmniRoute Remote") +**2. Create a new OAuth 2.0 Client ID** -**3. Legg til autoriserte omdirigerings-URIer** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -I feltet**«Authorized redirect URIs»**legger du til:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Erstatt `din-server.com` med serverens domene eller IP (inkluder porten om nødvendig, f.eks. `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. Lagre og kopier legitimasjonen** +After creating, Google will show the **Client ID** and **Client Secret**. -Etter oppretting vil Google vise**klient-ID**og**klienthemmelighet**. +**5. Set environment variables** -**5. Angi miljøvariabler** +In your `.env` (or Docker environment variables): -I `.env` (eller Docker-miljøvariabler):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1788,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Start OmniRoute**på nytt```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Prøv å koble til på nytt** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Dashboard → Leverandører → Antigravity (eller Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google vil nå omdirigere riktig til `https://din-server.com/callback`.--- +--- #### Temporary workaround (without custom credentials) -Hvis du ikke vil sette opp din egen legitimasjon akkurat nå, kan du fortsatt bruke den**manuelle URL-flyten**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute åpner Googles autorisasjons-URL -2. Etter godkjenning prøver Google å omdirigere til `localhost` (som mislykkes på den eksterne serveren) -3.**Kopiér hele nettadressen**fra nettleserens adressefelt (selv om siden ikke lastes inn) -4. Lim inn den URL-en i feltet vist i OmniRoute-tilkoblingsmodalen -5. Klikk på**"Koble til"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Dette fungerer fordi autorisasjonskoden i URL-en er gyldig uavhengig av om omdirigeringssiden er lastet.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. - -🇧🇷 Versão em Português#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Os testedores**Antigravity**og**Gemini CLI**usam**Google OAuth 2.0**for autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja**exatamente**uma das URIs pre-cadastradas no Google Cloud Console do aplicativo. +
+🇧🇷 Versão em Português -Som credenciais OAuth embutidas no OmniRoute estão cadastradas**apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (eks: `https://omniroute.meuservidor.com`), o Google avviser en autenticação com:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Você presa criar um**OAuth 2.0 Client ID**no Google Cloud Console com en URI for seu service.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Tilgang til Google Cloud Console** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) **2. Crie um novo OAuth 2.0 Client ID** -- Klikk dem**"+ Opprett legitimasjon"**→**"OAuth-klient-ID"** -- Tipo de aplicativo:**"Nettapplikasjon"** -- Navn: escolha qualquer nome (eks: `OmniRoute Remote`) +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Adicione som autorisert omdirigerings-URI** +**3. Adicione as Authorized Redirect URIs** -Ingen campo**"Authorized redirect URIs"**, adicione:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Substitua `seu-servidor.com` pelo domínio eller IP do seu servidor (inkludert en porta se necessário, eks: `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. Salve e copy as credenciais** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Após criar, o Google mostrará o**Client ID**e o**Client Secret**. +**5. Configure as variáveis de ambiente** -**5. Konfigurer som variáveis de ambiente** +No seu `.env` (ou nas variáveis de ambiente do Docker): -No seu `.env` (ou nas variáveis de ambiente do Docker):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1867,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reinicie o OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute - -```` +``` **7. Tente conectar novamente** -Dashboard → Leverandører → Antigravity (ou Gemini CLI) → OAuth +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` og autenticação funcionará.--- +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. + +--- #### Workaround temporário (sem configurar credenciais próprias) -Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo**manual de URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. O OmniRoute abrirá en URL fra autorização til Google -2. Após você autorizar, o Google tentará redirecionar for `localhost` (que falha no servidor remoto) -3.**Kopier en URL fullført**da barra de endereço do seu nettleseren (mesmo que a página não carregue) +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) 4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute -5. Klikk på**"Koble til"** +5. Clique em **"Connect"** -> Este workaround funciona porque or código de autorização na URL é válido independente do redirect ter carregado or não.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1905,64 +2171,73 @@ Se não quiser criar credenciais próprias agora, ainda é possível usar o flux ## 🛠️ Tech Stack - -Klikk for å utvide teknisk stackdetaljer +
+Click to expand tech stack details --**Kjøretid**: Node.js 18–22 LTS (⚠️ Node.js 24+ er**ikke støttet**— «better-sqlite3» native binærfiler er inkompatible) --**Språk**: TypeScript 5.9 —**100 % TypeScript**på tvers av `src/` og `open-sse/` (null `noen` i kjernemoduler siden v2.0) --**Rammeverk**: Next.js 16 + React 19 + Tailwind CSS 4 --**Database**: LowDB (JSON) + SQLite (domenetilstand + proxy-logger + MCP-revisjon + rutingbeslutninger) --**Skjemaer**: Zod (MCP-verktøy I/O-validering, API-kontrakter) --**Protokoller**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Strøming**: Server-sendte hendelser (SSE) --**Auth**: OAuth 2.0 (PKCE) + JWT + API-nøkler + MCP-omfanget autorisasjon --**Test**: Node.js testløper + Vitest (900+ tester inkludert enhet, integrasjon, E2E) --**CI/CD**: GitHub-handlinger (automatisk npm-publisering + Docker Hub ved utgivelse) --**Nettsted**: [omniroute.online](https://omniroute.online) --**Pakke**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Resiliens**: Strømbryter, eksponentiell backoff, anti-trdnende flokk, TLS-spoofing, auto-combo selvhelbredelse
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Dokumentasjon -| Dokument | Beskrivelse | -| ------------------------------------------------------ | ---------------------------------------------------------- | -| [Brukerveiledning](docs/USER_GUIDE.md) | Leverandører, kombinasjoner, CLI-integrasjon, distribusjon | -| [API-referanse](docs/API_REFERENCE.md) | Alle endepunkter med eksempler | -| [MCP-server](open-sse/mcp-server/README.md) | 16 MCP-verktøy, IDE-konfigurasjoner, Python/TS/Go-klienter | -| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0-protokoll, ferdigheter, streaming, oppgavehåndtering | -| [Auto-combo Engine](docs/auto-combo.md) | 6-faktor scoring, moduspakker, selvhelbredende | -| [Feilsøking](docs/TROUBLESHOOTING.md) | Vanlige problemer og løsninger | -| [Arkitektur](docs/ARCHITECTURE.md) | Systemarkitektur og innvendig | -| [Bidrar](CONTRIBUTING.md) | Utviklingsoppsett og retningslinjer | -| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0-spesifikasjon | -| [Sikkerhetspolicy](SECURITY.md) | Sårbarhetsrapportering og sikkerhetspraksis | -| [VM-distribusjon](docs/VM_DEPLOYMENT_GUIDE.md) | Komplett guide: VM + nginx + Cloudflare-oppsett | -| [Funksjonsgalleri](docs/FEATURES.md) | Visuell dashbordomvisning med skjermbilder | -| [Utgivelsessjekkliste](docs/RELEASE_CHECKLIST.md) | Valideringstrinn før utgivelse |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute har**210+ funksjoner planlagt**på tvers av flere utviklingsfaser. Her er nøkkelområdene: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Kategori | Planlagte funksjoner | Høydepunkter | -| ------------------------------ | ---------------- | ---------------------------------------------------------------------------------------------- | -| 🧠**Routing og intelligens**| 25+ | Ruting med lavest ventetid, tag-basert ruting, forhåndskontroll av kvoter, valg av P2C-konto | -| 🔒**Sikkerhet og overholdelse**| 20+ | SSRF-herding, tilsløring av legitimasjon, hastighetsgrense per endepunkt, styringsnøkkelomfang | -| 📊**Observerbarhet**| 15+ | OpenTelemetry-integrasjon, kvoteovervåking i sanntid, kostnadssporing per modell | -| 🔄**Tilbyderintegrasjoner**| 20+ | Dynamisk modellregister, leverandørnedkjøling, multi-konto Codex, Copilot-kvoteparsing | -| ⚡**Ytelse**| 15+ | Dobbelt hurtigbufferlag, promptbuffer, svarbuffer, streaming keepalive, batch API | -| 🌐**Økosystem**| 10+ | WebSocket API, config hot-reload, distribuert config store, kommersiell modus |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**OpenCode Integration**— Innebygd leverandørstøtte for OpenCode AI-kodings-IDE -- 🔗**TRAE-integrasjon**— Full støtte for utviklingsrammeverket for TRAE AI -- 📦**Batch API**— Asynkron batchbehandling for bulkforespørsler -- 🎯**Tag-basert ruting**— Ruteforespørsler basert på tilpassede tagger og metadata -- 💰**Laveste kostnadsstrategi**— Velg automatisk den billigste tilgjengelige leverandøren +### 🔜 Coming Soon -> 📝 Full funksjonsspesifikasjoner tilgjengelig i [`docs/new-features/`](docs/new-features/) (217 detaljerte spesifikasjoner)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1970,18 +2245,20 @@ OmniRoute har**210+ funksjoner planlagt**på tvers av flere utviklingsfaser. Her ### How to Contribute -1. Fordel depotet -2. Lag din funksjonsgren (`git checkout -b feature/amazing-feature`) -3. Overfør endringene dine (`git commit -m 'Legg til fantastisk funksjon'`) -4. Skyv til grenen (`git push origin feature/amazing-feature`) -5. Åpne en pull-forespørsel +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Se [CONTRIBUTING.md](CONTRIBUTING.md) for detaljerte retningslinjer.### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1993,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Spesiell takk til**[9router](https://github.com/decolua/9router)**av**[decolua](https://github.com/decolua)**– det originale prosjektet som inspirerte denne gaffelen. OmniRoute bygger på det utrolige grunnlaget med tilleggsfunksjoner, multimodale APIer og en full TypeScript-omskriving. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Spesiell takk til**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— den originale Go-implementeringen som inspirerte denne JavaScript-porten.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Lisens -MIT-lisens - se [LISENS](LISENS) for detaljer.--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/no/docs/ARCHITECTURE.md b/docs/i18n/no/docs/ARCHITECTURE.md index 74dcb3218b..187931d9fd 100644 --- a/docs/i18n/no/docs/ARCHITECTURE.md +++ b/docs/i18n/no/docs/ARCHITECTURE.md @@ -4,81 +4,93 @@ --- -_Sist oppdatert: 2026-03-28_## Executive Summary -OmniRoute er en lokal AI-rutinggateway og dashbord bygget på Next.js. -Den gir et enkelt OpenAI-kompatibelt endepunkt (`/v1/*`) og ruter trafikk på tvers av flere oppstrømsleverandører med oversettelse, reserve, token-oppdatering og brukssporing. -Kjernefunksjoner: +_Last updated: 2026-03-28_ -- OpenAI-kompatibel API-overflate for CLI/verktøy (28 leverandører) -- Forespørsel/svar oversettelse på tvers av leverandørformater -- Modellkombinasjonsfallback (multimodellsekvens) -- Reserveback på kontonivå (multikonto per leverandør) -- OAuth + API-nøkkelleverandør tilkoblingsadministrasjon -- Innebyggingsgenerering via `/v1/embeddings` (6 leverandører, 9 modeller) -- Bildegenerering via `/v1/images/generations` (4 leverandører, 9 modeller) -- Tenk tag-parsing (`...`) for resonneringsmodeller -- Respons sanitization for streng OpenAI SDK-kompatibilitet -- Rollenormalisering (utvikler→system, system→bruker) for kompatibilitet på tvers av leverandører -- Konvertering av strukturert utdata (json_schema → Gemini responseSchema) -- Lokal utholdenhet for leverandører, nøkler, aliaser, kombinasjoner, innstillinger, priser -- Bruks-/kostnadssporing og forespørselslogging -- Valgfri skysynkronisering for synkronisering av flere enheter/tilstander -- IP-godkjenningsliste/blokkeringsliste for API-tilgangskontroll -- Tenker budsjettstyring (gjennomgang/auto/tilpasset/tilpasset) -- Injeksjon av et globalt system -- Sesjonssporing og fingeravtrykk -- Forbedret prisbegrensning per konto med leverandørspesifikke profiler -- Strømbrytermønster for leverandørens motstandskraft -- Anti-tordenbeskyttelse med mutex-låsing -- Signaturbasert forespørselsdedupliseringsbuffer -- Domenelag: modelltilgjengelighet, kostnadsregler, reservepolicy, lockoutpolicy -- Vedvarende domenetilstand (SQLite-gjennomskrivingsbuffer for reserver, budsjetter, lockouts, strømbrytere) -- Policymotor for sentralisert forespørselsevaluering (lockout → budsjett → reserve) -- Be om telemetri med p50/p95/p99 latensaggregering -- Korrelasjons-ID (X-Request-Id) for ende-til-ende-sporing -- Overholdelsesrevisjonslogging med opt-out per API-nøkkel -- Eval rammeverk for LLM kvalitetssikring -- Resilience UI-dashbord med sanntids strømbryterstatus -- Modulære OAuth-leverandører (12 individuelle moduler under `src/lib/oauth/providers/`) +## Executive Summary -Primær kjøretidsmodell: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -– Next.js app-ruter under `src/app/api/*` implementerer både dashbord-APIer og kompatibilitets-APIer +Core capabilities: -- En delt SSE/rutingkjerne i `src/sse/*` + `open-sse/*` håndterer leverandørutførelse, oversettelse, streaming, fallback og bruk## Scope and Boundaries +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Lokal gateway kjøretid -- Dashboard management APIer -- Leverandørautentisering og tokenoppdatering -- Be om oversettelse og SSE-streaming -- Lokal stat + bruksutholdenhet -- Valgfri skysynkroniseringsorkestrering### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Implementering av skytjenester bak `NEXT_PUBLIC_CLOUD_URL` -- Leverandør SLA/kontrollplan utenfor lokal prosess -- Eksterne CLI-binærfiler i seg selv (Claude CLI, Codex CLI, etc.)## Dashboard Surface (Current) +### Out of Scope -Hovedsider under `src/app/(dashboard)/dashboard/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — hurtigstart + leverandøroversikt -- `/dashboard/endepunkt` — endepunktproxy + MCP + A2A + API-endepunktfaner -- `/dashboard/providers` — leverandørtilkoblinger og legitimasjon -- `/dashboard/combos` — kombinasjonsstrategier, maler, modellrutingsregler -- `/dashboard/costs` — kostnadsaggregering og prissynlighet -- `/dashboard/analytics` — bruksanalyse og evalueringer -- `/dashboard/limits` — kvote-/satskontroller -- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generering -- `/dashboard/agents` — oppdagede ACP-agenter + tilpasset agentregistrering -- `/dashboard/media` — bilde/video/musikklekeplass -- `/dashboard/search-tools` — testing og historikk for søkeleverandører -- `/dashboard/helse` — oppetid, strømbrytere, rategrenser -- `/dashboard/logs` — request/proxy/audit/console logger -- `/dashboard/innstillinger` — systeminnstillinger-faner (generelt, ruting, kombinasjonsstandarder, etc.) -- `/dashboard/api-manager` — API-nøkkellivssyklus og modelltillatelser## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -130,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Hovedkataloger: +Main directories: -- `src/app/api/v1/*` og `src/app/api/v1beta/*` for kompatibilitets-APIer -- `src/app/api/*` for administrasjons-/konfigurasjons-APIer -- Neste omskrives i `next.config.mjs` kart `/v1/*` til `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Viktige kompatibilitetsruter: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` – inkluderer egendefinerte modeller med `custom: true` -- `src/app/api/v1/embeddings/route.ts` — generering av innebygging (6 leverandører) -- `src/app/api/v1/images/generations/route.ts` — bildegenerering (4+ leverandører inkl. Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedikert chat per leverandør -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedikerte innbygginger per leverandør -- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedikerte bilder per leverandør +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` - `src/app/api/v1beta/models/[...path]/route.ts` -Administrasjonsdomener: +Management domains: -- Auth/innstillinger: `src/app/api/auth/*`, `src/app/api/settings/*` -- Leverandører/tilkoblinger: `src/app/api/providers*` -- Leverandørnoder: `src/app/api/provider-nodes*` -- Egendefinerte modeller: `src/app/api/provider-models` (GET/POST/DELETE) -- Modellkatalog: `src/app/api/models/route.ts` (GET) -- Proxy-konfigurasjon: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` - Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- Bruk: `src/app/api/usage/*` +- Usage: `src/app/api/usage/*` - Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` -- CLI-verktøyhjelpere: `src/app/api/cli-tools/*` -- IP-filter: `src/app/api/settings/ip-filter` (GET/PUT) -- Tenkebudsjett: `src/app/api/settings/thinking-budget` (GET/PUT) -- Systemmelding: `src/app/api/settings/system-prompt` (GET/PUT) -- Økter: `src/app/api/sessions` (GET) -- Satsgrenser: `src/app/api/rate-limits` (GET) -- Resiliens: `src/app/api/resilience` (GET/PATCH) – leverandørprofiler, kretsbryter, rategrensetilstand -- Resilience reset: `src/app/api/resilience/reset` (POST) — tilbakestill brytere + nedkjøling -- Bufferstatistikk: `src/app/api/cache/stats` (GET/DELETE) -- Modelltilgjengelighet: `src/app/api/models/availability` (GET/POST) -- Telemetri: `src/app/api/telemetry/summary` (GET) - – Budsjett: `src/app/api/usage/budget` (GET/POST) -- Reservekjeder: `src/app/api/fallback/chains` (GET/POST/DELETE) -- Samsvarsrevisjon: `src/app/api/compliance/audit-log` (GET) +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) - Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- Retningslinjer: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Policies: `src/app/api/policies` (GET/POST) -Hovedstrømningsmoduler: +## 2) SSE + Translation Core -- Oppføring: `src/sse/handlers/chat.ts` -- Kjerneorkestrering: `open-sse/handlers/chatCore.ts` -- Leverandørutførelsesadaptere: `open-sse/executors/*` -- Formatdeteksjon/leverandørkonfigurasjon: `open-sse/services/provider.ts` -- Modellparse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Kontoreservelogikk: `open-sse/services/accountFallback.ts` -- Oversettelsesregister: `open-sse/translator/index.ts` -- Strømtransformasjoner: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- Bruksutvinning/normalisering: `open-sse/utils/usageTracking.ts` -- Think tag-parser: `open-sse/utils/thinkTagParser.ts` -- Innebyggingsbehandler: `open-sse/handlers/embeddings.ts` -- Innebyggingsleverandørregister: `open-sse/config/embeddingRegistry.ts` -- Bildegenereringsbehandler: `open-sse/handlers/imageGeneration.ts` -- Bildeleverandørs register: `open-sse/config/imageRegistry.ts` +Main flow modules: + +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` - Response sanitization: `open-sse/handlers/responseSanitizer.ts` -- Rollenormalisering: `open-sse/services/roleNormalizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -Tjenester (forretningslogikk): +Services (business logic): -- Kontovalg/score: `open-sse/services/accountSelector.ts` +- Account selection/scoring: `open-sse/services/accountSelector.ts` - Context lifecycle management: `open-sse/services/contextManager.ts` -- Håndhevelse av IP-filter: `open-sse/services/ipFilter.ts` -- Sesjonssporing: `open-sse/services/sessionManager.ts` -- Be om deduplisering: `open-sse/services/signatureCache.ts` -- Systemprompt-injeksjon: `open-sse/services/systemPrompt.ts` -- Tenkende budsjettstyring: `open-sse/services/thinkingBudget.ts` -- Ruting av jokertegnmodell: `open-sse/services/wildcardRouter.ts` -- Satsgrenseadministrasjon: `open-sse/services/rateLimitManager.ts` -- Strømbryter: `open-sse/services/circuitBreaker.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -Domenelagsmoduler: +Domain layer modules: -- Modelltilgjengelighet: `src/lib/domain/modelAvailability.ts` -- Kostnadsregler/budsjetter: `src/lib/domain/costRules.ts` -- Reservepolicy: `src/lib/domain/fallbackPolicy.ts` +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` - Combo resolver: `src/lib/domain/comboResolver.ts` - Lockout policy: `src/lib/domain/lockoutPolicy.ts` -- Policymotor: `src/domain/policyEngine.ts` — sentralisert lockout → budsjett → reserveevaluering -- Feilkodekatalog: `src/lib/domain/errorCodes.ts` -- Forespørsels-ID: `src/lib/domain/requestId.ts` -- Tidsavbrudd for henting: `src/lib/domain/fetchTimeout.ts` -- Be om telemetri: `src/lib/domain/requestTelemetry.ts` -- Samsvar/revisjon: `src/lib/domain/compliance/index.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` - Eval runner: `src/lib/domain/evalRunner.ts` -- Vedvarende domenetilstand: `src/lib/db/domainState.ts` — SQLite CRUD for reservekjeder, budsjetter, kostnadshistorikk, lockout-tilstand, strømbrytere +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -OAuth-leverandørmoduler (12 individuelle filer under `src/lib/oauth/providers/`): +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -- Registerindeks: `src/lib/oauth/providers/index.ts` -- Individuelle leverandører: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts.`s.`s.`s.`s.`s.`s.` -- Tynn innpakning: `src/lib/oauth/providers.ts` — re-eksport fra individuelle moduler## 3) Persistence Layer +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -Primær tilstand DB (SQLite): +## 3) Persistence Layer -- Kjerneinfra: `src/lib/db/core.ts` (better-sqlite3, migreringer, WAL) -- Re-eksport fasade: `src/lib/localDb.ts` (tynt kompatibilitetslag for innringere) -- fil: `${DATA_DIR}/storage.sqlite` (eller `$XDG_CONFIG_HOME/omniroute/storage.sqlite` når angitt, ellers `~/.omniroute/storage.sqlite`) -- enheter (tabeller + KV-navnerom): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +Primary state DB (SQLite): -Bruksutholdenhet: +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -- fasade: `src/lib/usageDb.ts` (dekomponerte moduler i `src/lib/usage/*`) -- SQLite-tabeller i `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` -- valgfrie filartefakter forblir for kompatibilitet/feilsøking (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- eldre JSON-filer migreres til SQLite ved oppstartsmigreringer når de finnes +Usage persistence: -Domenetilstand DB (SQLite): +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- `src/lib/db/domainState.ts` — CRUD-operasjoner for domenetilstand -- Tabeller (opprettet i `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- Gjennomskrivingsbuffermønster: kart i minnet er autoritative under kjøring; mutasjoner skrives synkront til SQLite; tilstand gjenopprettes fra DB ved kaldstart## 4) Auth + Security Surfaces +Domain State DB (SQLite): -- Dashboard-informasjonskapselauth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- Generering/verifisering av API-nøkler: `src/shared/utils/apiKey.ts` -- Leverandørhemmeligheter vedvarte i oppføringer for 'providerConnections' -- Utgående proxy-støtte via `open-sse/utils/proxyFetch.ts` (env vars) og `open-sse/utils/networkProxy.ts` (konfigurerbar per leverandør eller global)## 5) Cloud Sync +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start -- Planlegger init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Periodisk oppgave: `src/shared/services/cloudSyncScheduler.ts` -- Periodisk oppgave: `src/shared/services/modelSyncScheduler.ts` -- Kontrollrute: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -339,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Reservebeslutninger er drevet av `open-sse/services/accountFallback.ts` ved bruk av statuskoder og feilmeldingsheuristikk. Kombinasjonsruting legger til en ekstra beskyttelse: 400-er med leverandøromfang som oppstrøms innholdsblokkering og rollevalideringsfeil blir behandlet som modelllokale feil, slik at senere kombinasjonsmål fortsatt kan kjøres.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -369,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -Oppdatering under live trafikk utføres inne i `open-sse/handlers/chatCore.ts` via eksekveren `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -401,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -Periodisk synkronisering utløses av 'CloudSyncScheduler' når skyen er aktivert.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -502,12 +532,14 @@ erDiagram } ``` -Fysiske lagringsfiler: +Physical storage files: -- primær kjøretids-DB: `${DATA_DIR}/storage.sqlite` -- be om logglinjer: `${DATA_DIR}/log.txt` (compat/debug artefakt) -- strukturerte anropsnyttelastarkiver: `${DATA_DIR}/call_logs/` -- valgfrie oversetter/forespørsler om feilsøkingsøkter: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -542,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: kompatibilitets-APIer -- `src/app/api/v1/providers/[provider]/*`: dedikerte ruter per leverandør (chat, innebygging, bilder) -- `src/app/api/providers*`: leverandør CRUD, validering, testing -- `src/app/api/provider-nodes*`: tilpasset kompatibel nodeadministrasjon -- `src/app/api/provider-models`: tilpasset modelladministrasjon (CRUD) -- `src/app/api/models/route.ts`: modellkatalog API (aliaser + tilpassede modeller) -- `src/app/api/oauth/*`: OAuth/enhetskode flyter -- `src/app/api/keys*`: lokal API-nøkkellivssyklus -- `src/app/api/models/alias`: aliasadministrasjon -- `src/app/api/combos*`: reservekombinasjonsadministrasjon -- `src/app/api/pricing`: prisoverstyringer for kostnadsberegning -- `src/app/api/settings/proxy`: proxy-konfigurasjon (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: utgående proxy-tilkoblingstest (POST) -- `src/app/api/usage/*`: API-er for bruk og logger -- `src/app/api/sync/*` + `src/app/api/cloud/*`: skysynkronisering og skyvendte hjelpere -- `src/app/api/cli-tools/*`: lokale CLI-konfigurasjonsforfattere/kontrollere -- `src/app/api/settings/ip-filter`: IP-godkjenningsliste/blokkeringsliste (GET/PUT) -- `src/app/api/settings/thinking-budget`: budsjettkonfigurasjon for tenketoken (GET/PUT) -- `src/app/api/settings/system-prompt`: global systemmelding (GET/PUT) -- `src/app/api/sessions`: aktiv øktoppføring (GET) -- `src/app/api/rate-limits`: rategrensestatus per konto (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: forespørsel om parse, kombinasjonshåndtering, kontovalgsløyfe -- `open-sse/handlers/chatCore.ts`: oversettelse, eksekutorutsendelse, prøv på nytt/oppdateringshåndtering, strømoppsett -- `open-sse/executors/*`: leverandørspesifikk nettverks- og formatatferd### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: oversetterregister og orkestrering -- Be om oversettere: `open-sse/translator/request/*` -- Responsoversettere: `open-sse/translator/response/*` -- Formatkonstanter: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: vedvarende config/state og domene persistens på SQLite -- `src/lib/localDb.ts`: re-eksport av kompatibilitet for DB-moduler -- `src/lib/usageDb.ts`: brukshistorikk/anropsloggfasade på toppen av SQLite-tabeller## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Hver leverandør har en spesialisert executor som utvider `BaseExecutor` (i `open-sse/executors/base.ts`), som gir URL-bygging, header-konstruksjon, forsøk på nytt med eksponentiell backoff, credential refresh hooks og `execute()`-orkestreringsmetoden. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Utfører | Leverandør(er) | Spesiell håndtering | -| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ---------------------------------------------------------------------------- | -| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamisk URL/header-konfigurasjon per leverandør | -| `AntigravityExecutor` | Google Antigravity | Egendefinerte prosjekt-/sesjons-ID-er, Prøv på nytt etter parsing | -| `CodexExecutor` | OpenAI Codex | Injiserer systeminstruksjoner, tvinger resonnementinnsats | -| `CursorExecutor` | Markør IDE | ConnectRPC-protokoll, Protobuf-koding, forespørsel om signering via sjekksum | -| `GithubExecutor` | GitHub Copilot | Copilot token oppdatering, VSCode-lignende overskrifter | -| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binært format → SSE-konvertering | -| `GeminiCLIExecutor` | Gemini CLI | Oppdateringssyklus for Google OAuth-token | +### Persistence -Alle andre leverandører (inkludert tilpassede kompatible noder) bruker "DefaultExecutor".## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Leverandør | Format | Auth | Stream | Ikke-stream | Token oppdatering | Bruks-API | -| ---------------- | --------------- | --------------------- | ---------------- | ----------- | ----------------- | ------------------------ | ------------------------------ | -| Claude | claude | API-nøkkel / OAuth | ✅ | ✅ | ✅ | ⚠️ Kun administrator | -| Tvillingene | Gemini | API-nøkkel / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | -| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | -| Antigravitasjon | antigravitasjon | OAuth | ✅ | ✅ | ✅ | ✅ Full kvote API | -| OpenAI | openai | API-nøkkel | ✅ | ✅ | ❌ | ❌ | -| Codex | openai-svar | OAuth | ✅ tvunget | ❌ | ✅ | ✅ Satsgrenser | -| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Kvote øyeblikksbilder | -| Markør | markør | Egendefinert sjekksum | ✅ | ✅ | ❌ | ❌ | -| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Bruksgrenser | -| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per forespørsel | -| Qoder | openai | OAuth (Grunnleggende) | ✅ | ✅ | ✅ | ⚠️ Per forespørsel | -| OpenRouter | openai | API-nøkkel | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | claude | API-nøkkel | ✅ | ✅ | ❌ | ❌ | -| DeepSeek | openai | API-nøkkel | ✅ | ✅ | ❌ | ❌ | -| Groq | openai | API-nøkkel | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | openai | API-nøkkel | ✅ | ✅ | ❌ | ❌ | -| Mistral | openai | API-nøkkel | ✅ | ✅ | ❌ | ❌ | -| Forvirring | openai | API-nøkkel | ✅ | ✅ | ❌ | ❌ | -| Sammen AI | openai | API-nøkkel | ✅ | ✅ | ❌ | ❌ | -| Fyrverkeri AI | openai | API-nøkkel | ✅ | ✅ | ❌ | ❌ | -| Cerebras | openai | API-nøkkel | ✅ | ✅ | ❌ | ❌ | -| Sammenheng | openai | API-nøkkel | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | openai | API-nøkkel | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Oppdagede kildeformater inkluderer: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. + +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | + +All other providers (including custom compatible nodes) use the `DefaultExecutor`. + +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: - `openai` -- `openai-svar` -- `Claude` -- 'tvilling' +- `openai-responses` +- `claude` +- `gemini` -Målformater inkluderer: +Target formats include: -- OpenAI chat/svar +- OpenAI chat/Responses - Claude -- Gemini/Gemini-CLI/Antigravity konvolutt +- Gemini/Gemini-CLI/Antigravity envelope - Kiro -- Markør +- Cursor -Oversettelser bruker**OpenAI som hub-format**– alle konverteringer går gjennom OpenAI som mellomliggende:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Oversettelser velges dynamisk basert på kildens nyttelastform og leverandørens målformat. +Additional processing layers in the translation pipeline: -Ytterligere behandlingslag i oversettelsespipelinen: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Sansering av svar**- Fjerner ikke-standardfelter fra svar i OpenAI-format (både strømming og ikke-strømming) for å sikre streng SDK-overholdelse --**Rollenormalisering**— Konverterer `utvikler` → `system` for ikke-OpenAI-mål; slår sammen `system` → `bruker` for modeller som avviser systemrollen (GLM, ERNIE) --**Tenk-tag-utvinning**— analyserer «...»-blokker fra innhold til «reasoning_content»-feltet --**Structured output**— Konverterer OpenAI `response_format.json_schema` til Geminis `responseMimeType` + `responseSchema`## Supported API Endpoints +## Supported API Endpoints -| Endepunkt | Format | Handler | -| ---------------------------------------------------------- | ------------------ | -------------------------------------------------------------------------- | -| `POST /v1/chat/fullføringer` | OpenAI Chat | `src/sse/handlers/chat.ts` | -| `POST /v1/meldinger` | Claude Meldinger | Samme behandler (automatisk oppdaget) | -| `POST /v1/responses` | OpenAI-svar | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | -| `GET /v1/embeddings` | Modellliste | API-rute | -| `POST /v1/bilder/generasjoner` | OpenAI-bilder | `open-sse/handlers/imageGeneration.ts` | -| `GET /v1/images/generations` | Modellliste | API-rute | -| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedikert per leverandør med modellvalidering | -| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedikert per leverandør med modellvalidering | -| `POST /v1/providers/{provider}/images/generations` | OpenAI-bilder | Dedikert per leverandør med modellvalidering | -| `POST /v1/messages/count_tokens` | Claude Token Count | API-rute | -| `GET /v1/modeller` | OpenAI-modellliste | API-rute (chat + innebygging + bilde + tilpassede modeller) | -| `GET /api/modeller/katalog` | Katalog | Alle modeller gruppert etter leverandør + type | -| `POST /v1beta/models/*:streamGenerateContent` | Gemini innfødt | API-rute | -| `GET/PUT/DELETE /api/settings/proxy` | Proxy-konfigurasjon | Nettverks proxy-konfigurasjon | -| `POST /api/settings/proxy/test` | Proxy-tilkobling | Proxy helse/tilkoblingstestendepunkt | -| `GET/POST/DELETE /api/provider-models` | Leverandørmodeller | Leverandørmodellmetadata støtter tilpassede og administrerte tilgjengelige modeller |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -Bypass-behandleren (`open-sse/utils/bypassHandler.ts`) avskjærer kjente "kasting"-forespørsler fra Claude CLI – oppvarmingspinger, tittelutdrag og tokentellinger – og returnerer et**falsk svar**uten å forbruke oppstrømsleverandørtokens. Dette utløses bare når `User-Agent` inneholder `claude-cli`.## Request Logger Pipeline +## Bypass Handler -Forespørselsloggeren (`open-sse/utils/requestLogger.ts`) gir en 7-trinns debug logging pipeline, deaktivert som standard, aktivert via `ENABLE_REQUEST_LOGS=true`:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Filer skrives til `/logs//` for hver forespørselsøkt.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- Nedkjøling av leverandørens konto på forbigående/rate/auth-feil -- kontoreserve før mislykket forespørsel -- combo modell fallback når gjeldende modell/leverandørbane er oppbrukt## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- forhåndssjekk og oppdater med nytt forsøk for leverandører som kan oppdateres -- 401/403 prøv på nytt etter oppdateringsforsøk i kjernebanen## 3) Stream Safety +## 2) Token Expiry -- frakoblingsbevisst strømkontroller -- oversettelsesstrøm med flush ved slutten av strømmen og "[FERDIG]"-håndtering - – fallback for bruksestimat når leverandørbruksmetadata mangler## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- Synkroniseringsfeil dukker opp, men lokal kjøretid fortsetter -- planleggeren har logikk som kan forsøke på nytt, men periodisk kjøring kaller for øyeblikket enkeltforsøkssynkronisering som standard## 5) Data Integrity +## 3) Stream Safety -- SQLite-skjemamigrering og auto-oppgraderingshooks ved oppstart -- eldre JSON → SQLite-migreringskompatibilitetsbane## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Synlighetskilder for kjøretid: +## 4) Cloud Sync Degradation -- konsolllogger fra `src/sse/utils/logger.ts` -- bruksaggregater per forespørsel i SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- fire-trinns detaljert nyttelastfangst i SQLite (`request_detail_logs`) når `settings.detailed_logs_enabled=true` -- statuslogg for tekstforespørsel i `log.txt` (valgfritt/kompat) -- valgfrie dype forespørsels-/oversettelseslogger under `logger/` når `ENABLE_REQUEST_LOGS=true` -- endepunkter for dashbordbruk (`/api/usage/*`) for brukergrensesnittforbruk +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -Detaljert forespørsel om nyttelastfangst lagrer opptil fire JSON-nyttelaststadier per rutet samtale: +## 5) Data Integrity -- rå forespørsel mottatt fra klienten -- oversatt forespørsel faktisk sendt oppstrøms -- leverandørsvar rekonstruert som JSON; streamede svar komprimeres til det endelige sammendraget pluss strømmetadata -- endelig kundesvar returnert av OmniRoute; streamede svar lagres i samme kompakte sammendragsskjema## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- JWT secret (`JWT_SECRET`) sikrer bekreftelse/signering av informasjonskapsler for dashbordøkten -- Oppstartsoppstart for passord ('INITIAL_PASSWORD') bør eksplisitt konfigureres for førstegangsklargjøring -- API-nøkkel HMAC-hemmelighet (`API_KEY_SECRET`) sikrer generert lokalt API-nøkkelformat -- Leverandørhemmeligheter (API-nøkler/-tokens) er bevart i lokal DB og bør beskyttes på filsystemnivå -- Sluttpunkter for skysynkronisering er avhengige av API-nøkkelautentisering + maskin-ID-semantikk## Environment and Runtime Matrix +## Observability and Operational Signals -Miljøvariabler som brukes aktivt av kode: +Runtime visibility sources: + +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption + +Detailed request payload capture stores up to four JSON payload stages per routed call: + +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: - App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` -- Lagring: `DATA_DIR` -- Kompatibel nodeoppførsel: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Valgfri lagringsbaseoverstyring (Linux/macOS når `DATA_DIR` er deaktivert): `XDG_CONFIG_HOME` -- Sikkerhetshashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` - Logging: `ENABLE_REQUEST_LOGS` -- Synkronisering/nettadresse i nettskyen: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- Utgående proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` og varianter av små bokstaver -- SOCKS5-funksjonsflagg: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` - – Plattform-/kjøretidshjelpere (ikke appspesifikk konfigurasjon): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` -1. `usageDb` og `localDb` deler samme grunnkatalogpolicy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) med eldre filmigrering. -2. `/api/v1/route.ts` delegerer til den samme enhetlige katalogbyggeren som brukes av `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) for å unngå semantisk drift. -3. Forespørselslogger skriver fullstendige overskrifter/tekst når den er aktivert; behandle loggkatalogen som sensitiv. -4. Skyadferd avhenger av korrekt `NEXT_PUBLIC_BASE_URL` og tilgjengelighet for skyendepunkter. -5. `open-sse/`-katalogen er publisert som `@omniroute/open-sse`**npm-arbeidsområdepakken**. Kildekoden importerer den via `@omniroute/open-sse/...` (løst av Next.js `transpilePackages`). Filbaner i dette dokumentet bruker fortsatt katalognavnet `open-sse/` for konsistens. -6. Diagrammer i dashbordet bruker**Recharts**(SVG-basert) for tilgjengelige, interaktive analysevisualiseringer (stolpediagram for modellbruk, leverandøroversiktstabeller med suksessrater). -7. E2E-tester bruker**Playwright**(`tests/e2e/`), kjøres via `npm run test:e2e`. Enhetstester bruker**Node.js testløper**(`tests/unit/`), kjøres via `npm run test:unit`. Kildekoden under `src/` er**TypeScript**(`.ts`/`.tsx`); «open-sse/»-arbeidsområdet forblir JavaScript (.js). -8. Innstillinger-siden er organisert i 5 faner: Sikkerhet, Ruting (6 globale strategier: fill-first, round-robin, p2c, random, minst brukt, kostnadsoptimalisert), Resiliens (redigerbare hastighetsgrenser, strømbryter, policyer), AI (tenkebudsjett, systemprompt, promptbuffer), Advanced (proxy).## Operational Verification Checklist +## Known Architectural Notes -- Bygg fra kilden: `npm run build` -- Bygg Docker-bilde: `docker build -t omniroute .` -- Start tjenesten og bekreft: -- `GET /api/innstillinger` -- `GET /api/v1/modeller` -- CLI-målbase-URL skal være «http://:20128/v1» når «PORT=20128» +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` +- `GET /api/v1/models` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/no/docs/FEATURES.md b/docs/i18n/no/docs/FEATURES.md index f39905d00d..97796e5573 100644 --- a/docs/i18n/no/docs/FEATURES.md +++ b/docs/i18n/no/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Visuell veiledning til hver del av OmniRoute-dashbordet.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Administrer AI-leverandørtilkoblinger: OAuth-leverandører (Claude Code, Codex, Gemini CLI), API-nøkkelleverandører (Groq, DeepSeek, OpenRouter) og gratisleverandører (Qoder, Qwen, Kiro). Kiro-kontoer inkluderer sporing av kredittsaldo – gjenværende kreditter, total kvote og fornyelsesdato synlig i Dashboard → Bruk.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Lag modellrutingskombinasjoner med 6 strategier: prioritet, vektet, round-robin, tilfeldig, minst brukt og kostnadsoptimalisert. Hver kombinasjon kjeder flere modeller med automatisk fallback og inkluderer raske maler og beredskapskontroller.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Omfattende bruksanalyse med symbolforbruk, kostnadsestimater, aktivitetsvarmekart, ukentlige distribusjonsdiagrammer og sammenbrudd per leverandør.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Sanntidsovervåking: oppetid, minne, versjon, latenspersentiler (p50/p95/p99), hurtigbufferstatistikk og leverandørens strømbrytertilstander.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Fire moduser for feilsøking av API-oversettelser:**Lekeplass**(formatkonvertering),**Chattester**(liveforespørsler),**Testbenk**(batch-tester) og**Live Monitor**(sanntidsstrøm).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Test hvilken som helst modell direkte fra dashbordet. Velg leverandør, modell og endepunkt, skriv forespørsler med Monaco Editor, strøm svar i sanntid, avbryt midtstrøm og se tidsberegninger.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Tilpassbare fargetemaer for hele dashbordet. Velg mellom 7 forhåndsinnstilte farger (korall, blå, rød, grønn, fiolett, oransje, cyan) eller lag et tilpasset tema ved å velge en sekskantfarge. Støtter lys, mørk og systemmodus.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Omfattende innstillingspanel med faner: +Comprehensive settings panel with tabs: --**Generelt**— Systemlagring, sikkerhetskopiering (eksport/import database) -**Utseende**— Temavelger (mørkt/lys/system), forhåndsinnstillinger for fargetema og egendefinerte farger, synlighet av helselogg, synlighetskontroller for sidefeltelementer -**Sikkerhet**— API-endepunktbeskyttelse, tilpasset leverandørblokkering, IP-filtrering, øktinformasjon -**Routing**— Modellaliaser, forringelse av bakgrunnsoppgaver -**Resiliens**— Utholdenhet for frekvensgrense, innstilling av strømbryter, automatisk deaktivering av utestengte kontoer, overvåking av leverandørens utløp -**Avansert**— Konfigurasjonsoverstyringer, konfigurasjonsrevisjonsspor, fallback-degraderingsmodus![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Ett-klikks konfigurasjon for AI-kodingsverktøy: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor og Factory Droid. Inneholder automatisk konfigurering/tilbakestilling, tilkoblingsprofiler og modellkartlegging.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Dashboard for å oppdage og administrere CLI-agenter. Viser et rutenett med 14 innebygde agenter (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) med: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Installasjonsstatus**— Installert / Ikke funnet med versjonsdeteksjon -**Protokollmerker**— stdio, HTTP osv. -**Egendefinerte agenter**- Registrer et hvilket som helst CLI-verktøy via skjema (navn, binær, versjonskommando, spawn args) -**CLI Fingerprint Matching**— Veksle per leverandør for å matche native CLI-forespørselssignaturer, reduserer utestengelsesrisikoen samtidig som proxy-IP bevares--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Generer bilder, videoer og musikk fra dashbordet. Støtter OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open og MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Forespørselslogging i sanntid med filtrering etter leverandør, modell, konto og API-nøkkel. Viser statuskoder, tokenbruk, ventetid og svardetaljer.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Ditt enhetlige API-endepunkt med funksjonsoversikt: Chatfullføringer, Responses API, Innebygginger, Bildegenerering, Omrangering, Lydtranskripsjon, Tekst-til-tale, Moderasjoner og registrerte API-nøkler. Cloudflare Quick Tunnel-integrasjon og cloud proxy-støtte for ekstern tilgang.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -Opprett, omfang og tilbakekall API-nøkler. Hver nøkkel kan begrenses til spesifikke modeller/leverandører med full tilgang eller skrivebeskyttet tillatelse. Visuell nøkkelhåndtering med brukssporing.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Administrativ handlingssporing med filtrering etter handlingstype, aktør, mål, IP-adresse og tidsstempel. Full sikkerhetshendelseshistorikk.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Native Electron desktop-app for Windows, macOS og Linux. Kjør OmniRoute som en frittstående applikasjon med systemstatusfeltintegrasjon, offline-støtte, automatisk oppdatering og ett-klikks installering. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Nøkkelfunksjoner: +Key features: -- Avstemning av serverberedskap (ingen blank skjerm ved kaldstart) -- Systemstatusfelt med portadministrasjon -- Innholdssikkerhetspolicy -- Enkeltinstanslås -- Automatisk oppdatering ved omstart -- Plattformbetinget brukergrensesnitt (macOS trafikklys, Windows/Linux standard tittellinje) -- Herdet Electron build-emballasje – symboliserte "node_modules" i den frittstående pakken blir oppdaget og avvist før pakking, og forhindrer kjøretidsavhengighet av byggemaskinen (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Se [`electron/README.md`](../electron/README.md) for full dokumentasjon. +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/no/docs/TROUBLESHOOTING.md b/docs/i18n/no/docs/TROUBLESHOOTING.md index c61d8d273b..d2adcbf47f 100644 --- a/docs/i18n/no/docs/TROUBLESHOOTING.md +++ b/docs/i18n/no/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -Vanlige problemer og løsninger for OmniRoute.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Problem | Løsning | -| ---------------------------------------- | -------------------------------------------------------------------- | --- | -| Første pålogging fungerer ikke | Sett `INITIAL_PASSWORD` i `.env` (ingen hardkodet standard) | -| Dashboard åpnes på feil port | Sett `PORT=20128` og `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Ingen forespørselslogger under `logger/` | Sett `ENABLE_REQUEST_LOGS=true` | -| EACCES: tillatelse nektet | Sett `DATA_DIR=/path/to/writable/dir` for å overstyre `~/.omniroute` | -| Rutingstrategi lagrer ikke | Oppdater til v1.4.11+ (Zod-skjemafiks for varighet av innstillinger) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Årsak:**Leverandørkvoten er oppbrukt. +**Cause:** Provider quota exhausted. -**Fiks:** +**Fix:** -1. Sjekk dashbordkvotesporing -2. Bruk en kombinasjon med reservelag -3. Bytt til billigere/gratis lag### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Årsak:**Abonnementskvoten er oppbrukt. +### Rate Limiting -**Fiks:** +**Cause:** Subscription quota exhausted. -- Legg til reserve: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Bruk GLM/MiniMax som billig backup### OAuth Token Expired +**Fix:** -OmniRoute oppdaterer tokens automatisk. Hvis problemene vedvarer: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Dashboard → Leverandør → Koble til på nytt -2. Slett og legg til leverandørtilkoblingen på nytt--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Bekreft at «BASE_URL» peker til den kjørende forekomsten din (f.eks. «http://localhost:20128») -2. Bekreft "CLOUD_URL" peker til skyendepunktet ditt (f.eks. "https://omniroute.dev") -3. Hold `NEXT_PUBLIC_*`-verdier på linje med verdiene på tjenersiden### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Symptom:**`Uventet token 'd'...` på skyendepunkt for samtaler som ikke strømmer. +### Cloud `stream=false` Returns 500 -**Årsak:**Oppstrøms returnerer SSE-nyttelast mens klienten forventer JSON. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Løsning:**Bruk 'stream=true' for direkteanrop i skyen. Lokal kjøretid inkluderer SSE→JSON reserve.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Lag en ny nøkkel fra lokalt dashbord (`/api/keys`) -2. Kjør skysynkronisering: Aktiver Cloud → Synkroniser nå -3. Gamle/ikke-synkroniserte nøkler kan fortsatt returnere '401' på skyen--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Sjekk kjøretidsfeltene: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. For bærbar modus: bruk bildemål "runner-cli" (medfølgende CLI-er) -3. For vertsmonteringsmodus: sett `CLI_EXTRA_PATHS` og monter vertsbin-katalogen som skrivebeskyttet -4. Hvis `installed=true` og `runnable=false`: binær ble funnet, men mislyktes i helsesjekken### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Sjekk bruksstatistikk i Dashboard → Bruk -2. Bytt primærmodell til GLM/MiniMax -3. Bruk gratis nivå (Gemini CLI, Qoder) for ikke-kritiske oppgaver -4. Angi kostnadsbudsjetter per API-nøkkel: Dashboard → API-nøkler → Budsjett--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Sett `ENABLE_REQUEST_LOGS=true` i `.env`-filen. Logger vises under katalogen `logger/`.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Hovedtilstand: `${DATA_DIR}/storage.sqlite` (leverandører, kombinasjoner, aliaser, nøkler, innstillinger) -- Bruk: SQLite-tabeller i `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + valgfrie `${DATA_DIR}/log.txt` og `${DATA_DIR}/call_logs/` -- Forespørselslogger: `/logs/...` (når `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Når en leverandørs strømbryter er ÅPEN, blokkeres forespørsler til nedkjølingen utløper. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Fiks:** +**Fix:** -1. Gå til**Dashboard → Innstillinger → Resiliens** -2. Sjekk kretsbryterkortet for den berørte leverandøren -3. Klikk på**Tilbakestill alle**for å fjerne alle brytere, eller vent til nedkjølingen utløper -4. Bekreft at leverandøren faktisk er tilgjengelig før du tilbakestiller### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Hvis en leverandør gjentatte ganger går inn i ÅPEN tilstand: +### Provider keeps tripping the circuit breaker -1. Sjekk**Dashboard → Helse → Leverandørhelse**for feilmønsteret -2. Gå til**Innstillinger → Resiliens → Leverandørprofiler**og øk feilterskelen -3. Sjekk om leverandøren har endret API-grenser eller krever re-autentisering -4. Se gjennom latenstidstelemetri – høy latenstid kan forårsake timeout-baserte feil--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Sørg for at du bruker riktig prefiks: `deepgram/nova-3` eller `assemblyai/best` -- Bekreft at leverandøren er tilkoblet i**Dashboard → Leverandører**### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Sjekk støttede lydformater: "mp3", "wav", "m4a", "flac", "ogg", "webm" -- Bekreft at filstørrelsen er innenfor leverandørens grenser (vanligvis < 25 MB) -- Sjekk gyldigheten av leverandørens API-nøkkel i leverandørkortet--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Bruk**Dashboard → Oversetter**for å feilsøke problemer med formatoversettelse: +Use **Dashboard → Translator** to debug format translation issues: -| Modus | Når skal du bruke | -| ---------------- | ----------------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Lekeplass** | Sammenlign input/output formater side ved side — lim inn en mislykket forespørsel for å se hvordan den oversettes | -| **Chattetester** | Send direktemeldinger og inspiser hele nyttelasten for forespørsel/svar inkludert overskrifter | -| **Testbenk** | Kjør batch-tester på tvers av formatkombinasjoner for å finne hvilke oversettelser som er ødelagte | -| **Live Monitor** | Se forespørselsflyt i sanntid for å fange opp periodiske oversettelsesproblemer | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Tenkekoder vises ikke**— Sjekk om målleverandøren støtter tenkning og innstillingen av tenkebudsjettet -**Verktøyanrop dropper**— Noen formatoversettelser kan fjerne felt som ikke støttes; verifisere i Playground-modus -**Systemmelding mangler**— Claude og Gemini håndterer systemmeldinger annerledes; sjekk oversettelsen -**SDK returnerer rå streng i stedet for objekt**— Rettet i v1.1.0: svarrenser fjerner nå ikke-standard felt (`x_groq`, `usage_breakdown` osv.) som forårsaker OpenAI SDK Pydantic valideringsfeil -**GLM/ERNIE avviser 'system'-rollen**— Rettet i v1.1.0: rollenormalisering slår automatisk sammen systemmeldinger til brukermeldinger for inkompatible modeller -**`utviklerrollen gjenkjennes ikke**– Rettet i v1.1.0: automatisk konvertert til `system` for ikke-OpenAI-leverandører -**`json_schema` fungerer ikke med Gemini**- Rettet i v1.1.0: `response_format` er nå konvertert til Geminis `responseMimeType` + `responseSchema`--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- Automatisk takstgrense gjelder bare API-nøkkelleverandører (ikke OAuth/abonnement) -- Bekreft at**Innstillinger → Resiliens → Leverandørprofiler**har aktivert automatisk satsgrense -- Sjekk om leverandøren returnerer '429'-statuskoder eller 'Retry-After'-overskrifter### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -Leverandørprofiler støtter disse innstillingene: +### Tuning exponential backoff --**Basisforsinkelse**— Innledende ventetid etter første feil (standard: 1 s) -**Maksimal forsinkelse**— Maksimal ventetid (standard: 30s) -**Multiplikator**— Hvor mye skal forsinkelsen økes per påfølgende feil (standard: 2x)### Anti-thundering herd +Provider profiles support these settings: -Når mange samtidige forespørsler treffer en hastighetsbegrenset leverandør, bruker OmniRoute mutex + automatisk hastighetsbegrensning for å serialisere forespørsler og forhindre kaskadefeil. Dette er automatisk for API-nøkkelleverandører.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Noen OmniRoute-brukere plasserer gatewayen foran RAG- eller agentstabler. I disse oppsettene er det vanlig å se et merkelig mønster: OmniRoute ser sunt ut (leverandører oppe, ruteprofiler ok, ingen varsler om takstgrense), men det endelige svaret er fortsatt feil. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -I praksis kommer disse hendelsene vanligvis fra nedstrøms RAG-rørledningen, ikke fra selve gatewayen. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Hvis du vil ha et delt vokabular for å beskrive disse feilene, kan du bruke WFGY ProblemMap, en ekstern MIT-lisenstekstressurs som definerer seksten tilbakevendende RAG / LLM-feilmønstre. På et høyt nivå dekker det: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- gjenfinningsdrift og brutte kontekstgrenser -- tomme eller foreldede indekser og vektorlagre -- embedding versus semantisk mismatch -- Spørsmål om montering og kontekstvindu -- logisk kollaps og oversikre svar -- svikt i lang kjede og agentkoordinering -- multiagent minne og rolledrift -- problemer med distribusjon og bootstrap-bestilling +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -Ideen er enkel: +The idea is simple: -1. Når du undersøker et dårlig svar, fange opp: - - brukeroppgave og forespørsel - - rute eller leverandørkombinasjon i OmniRoute - - enhver RAG-kontekst brukt nedstrøms (hentede dokumenter, verktøyanrop, etc) -2. Kartlegg hendelsen til ett eller to WFGY ProblemMap-nummer (`No.1` … `No.16`). -3. Lagre nummeret i ditt eget dashbord, runbook eller hendelsessporing ved siden av OmniRoute-loggene. -4. Bruk den tilsvarende WFGY-siden til å bestemme om du må endre RAG-stack, retriever eller rutingstrategi. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -Fulltekst og konkrete oppskrifter live her (MIT-lisens, kun tekst): +Full text and concrete recipes live here (MIT license, text only): [WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -Du kan ignorere denne delen hvis du ikke kjører RAG eller agentpipelines bak OmniRoute.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**GitHub-problemer**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Architecture**: Se [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for interne detaljer -**API-referanse**: Se [`docs/API_REFERENCE.md`](API_REFERENCE.md) for alle endepunkter -**Helse Dashboard**: Sjekk**Dashboard → Health**for sanntids systemstatus -**Oversetter**: Bruk**Dashboard → Oversetter**for å feilsøke formatproblemer +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/no/llm.txt b/docs/i18n/no/llm.txt new file mode 100644 index 0000000000..02cb58b984 --- /dev/null +++ b/docs/i18n/no/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Norsk) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Oversikt + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Sikkerhet +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/phi/README.md b/docs/i18n/phi/README.md index 29887ecc78..5754f04f93 100644 --- a/docs/i18n/phi/README.md +++ b/docs/i18n/phi/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Ang iyong universal API proxy — isang endpoint, 60+ provider, zero downtime. Ngayon ay may**MCP Server (25 tool)**,**A2A Protocol**,**Memory/Skills System**at**Electron Desktop App**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Mga Pagkumpleto sa Chat • Mga Pag-embed • Pagbuo ng Imahe • Video • Musika • Audio • Pag-rerank •**Paghahanap sa Web**• MCP Server • A2A Protocol • 100% TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Ang iyong universal API proxy — isang endpoint, 60+ provider, zero downtime. [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Available sa:**🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,28 +60,30 @@ _Ang iyong universal API proxy — isang endpoint, 60+ provider, zero downtime. ## 📸 Dashboard Preview - -I-click upang makita ang mga screenshot ng dashboard +
+Click to see dashboard screenshots -| Pahina | Screenshot | -| ----------------------- | ------------------------------------------------- | ---------- | -| **Mga Provider** | ![Providers](docs/screenshots/01-providers.png) | -| **Combos** | ![Combos](docs/screenshots/02-combos.png) | -| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | -| **Kalusugan** | ![Health](docs/screenshots/04-health.png) | -| **Tagasalin** | ![Translator](docs/screenshots/05-translator.png) | -| **Mga Setting** | ![Mga Setting](docs/screenshots/06-settings.png) | -| **Mga CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | -| **Mga Log ng Paggamit** | ![Paggamit](docs/screenshots/08-usage.png) | -| **Mga Endpoint** | ![Mga Endpoint](docs/screenshots/09-endpoint.png) |
| +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | + + --- ### 🤖 Free AI Provider for your favorite coding agents -_Ikonekta ang anumang AI-powered IDE o CLI tool sa pamamagitan ng OmniRoute — libreng API gateway para sa walang limitasyong coding._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ - +
@@ -151,455 +160,529 @@ _Ikonekta ang anumang AI-powered IDE o CLI tool sa pamamagitan ng OmniRoute —
-📡 Kumokonekta ang lahat ng ahente sa pamamagitan ng http://localhost:20128/v1 o http://cloud.omniroute.online/v1 — isang config, walang limitasyong mga modelo at quota--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Itigil ang pag-aaksaya ng pera at pag-abot sa mga limitasyon:** +**Stop wasting money and hitting limits:** -- Ang quota ng subscription ay mag-e-expire nang hindi nagamit bawat buwan -- Pinipigilan ka ng mga limitasyon sa rate sa mid-coding -- Mga Mamahaling API ($20-50/buwan bawat provider) -- Manu-manong paglipat sa pagitan ng mga provider +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**Sumalutas ito ng OmniRoute:** +**OmniRoute solves this:** -- ✅**I-maximize ang mga subscription**- Subaybayan ang quota, gamitin ang bawat bit bago i-reset -- ✅**Auto fallback**- Subscription → API Key → Mura → Libre, zero downtime -- ✅**Multi-account**- Round-robin sa pagitan ng mga account sa bawat provider -- ✅**Universal**- Gumagana sa Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, anumang CLI tool--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Sumali sa aming komunidad!**[WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Humingi ng tulong, magbahagi ng mga tip, at manatiling updated. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Website**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Mga Isyu**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Grupo ng Komunidad](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Nag-aambag**: Tingnan ang [CONTRIBUTING.md](CONTRIBUTING.md), magbukas ng PR, o pumili ng `magandang unang isyu` -**Orihinal na Proyekto**: [9router by decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Kapag nagbubukas ng isyu, mangyaring patakbuhin ang system-info command at ilakip ang nabuong file:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Bumubuo ito ng `system-info.txt` gamit ang iyong bersyon ng Node.js, bersyon ng OmniRoute, mga detalye ng OS, mga naka-install na CLI tool (qoder, gemini, claude, codex, antigravity, droid, atbp.), status ng Docker/PM2, at mga system package — lahat ng kailangan namin para mabilis na mai-reproduce ang iyong isyu. Direktang ilakip ang file sa iyong isyu sa GitHub.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Ang bawat developer na gumagamit ng mga tool ng AI ay nahaharap sa mga problemang ito araw-araw.**Binuo ang OmniRoute para lutasin ang lahat ng ito — mula sa mga pag-overrun sa gastos hanggang sa mga panrehiyong bloke, mula sa mga sirang daloy ng OAuth hanggang sa mga pagpapatakbo ng protocol at pagmamasid sa enterprise. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. - -💸 1. "Nagbabayad ako para sa isang mamahaling subscription ngunit naaantala pa rin ng mga limitasyon" +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Nagbabayad ang mga developer ng $20–200/buwan para sa Claude Pro, Codex Pro, o GitHub Copilot. Kahit na nagbabayad, may kisame ang quota — 5h ng paggamit, lingguhang limitasyon, o bawat minutong limitasyon sa rate. Sesyon sa kalagitnaan ng coding, hihinto sa pagtugon ang provider at nawawalan ng daloy at pagiging produktibo ang developer. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Paano ito niresolba ng OmniRoute:** +**How OmniRoute solves it:** --**Smart 4-Tier Fallback**— Kung maubusan ang quota ng subscription, awtomatikong magre-redirect sa API Key → Murang → Libre nang walang manu-manong interbensyon --**Pagsubaybay sa Mga Limitasyon ng Provider**— Nagre-refresh ang mga naka-cache na snapshot ng quota sa isang iskedyul sa panig ng server (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) na may manual na pag-refresh na available sa UI --**Multi-Account Support**— Maramihang account sa bawat provider na may auto round-robin — kapag naubos ang isa, lilipat sa susunod --**Custom Combos**— Nako-customize na fallback chain na may 9 na diskarte sa pagbabalanse (priyoridad, weighted, fill-first, round-robin, P2C, random, hindi gaanong ginagamit, cost-optimized, strict-random) --**Codex Business Quotas**— Direktang pagsubaybay sa quota ng workspace ng Negosyo/Team sa dashboard
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard - -🔌 2. "Kailangan kong gumamit ng maramihang provider ngunit may iba't ibang API ang bawat isa" + -Gumagamit ang OpenAI ng isang format, gumagamit si Claude (Anthropic) ng isa pa, isa pa ang Gemini. Kung gusto ng isang dev na subukan ang mga modelo mula sa iba't ibang provider o fallback sa pagitan nila, kailangan nilang i-configure muli ang mga SDK, baguhin ang mga endpoint, harapin ang mga hindi tugmang format. Ang mga custom na provider (FriendLI, NIM) ay may hindi karaniwang mga endpoint ng modelo. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Paano ito niresolba ng OmniRoute:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Pinag-isang Endpoint**— Isang `http://localhost:20128/v1` ang nagsisilbing proxy para sa lahat ng 60+ provider --**Format Translation**— Awtomatiko at transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API --**Response Sanitization**— Tinatanggal ang mga hindi karaniwang field (`x_groq`, `usage_breakdown`, `service_tier`) na sumisira sa OpenAI SDK v1.83+ --**Role Normalization**— Kino-convert ang `developer` → `system` para sa mga provider na hindi OpenAI; `system` → `user` para sa GLM/ERNIE --**Think Tag Extraction**— Kinukuha ang `` block mula sa mga modelo tulad ng DeepSeek R1 tungo sa standardized na `reasoning_content` --**Structured Output para sa Gemini**— `json_schema` → `responseMimeType`/`responseSchema` awtomatikong conversion --**Nagde-default ang `stream` sa `false`**— Naka-align sa spec ng OpenAI, iniiwasan ang hindi inaasahang SSE sa mga Python/Rust/Go SDK
+**How OmniRoute solves it:** - -🌐 3. "Bina-block ng aking AI provider ang aking rehiyon/bansa" +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Hinaharang ng mga provider tulad ng OpenAI/Codex ang pag-access mula sa ilang partikular na heyograpikong rehiyon. Nakakakuha ang mga user ng mga error tulad ng `unsupported_country_region_territory` sa panahon ng OAuth at API na mga koneksyon. Ito ay lalo na nakakabigo para sa mga developer mula sa pagbuo ng mga bansa. + -**Paano ito niresolba ng OmniRoute:** +
+🌐 3. "My AI provider blocks my region/country" --**3-Level Proxy Config**— Nako-configure na proxy sa 3 antas: global (lahat ng trapiko), bawat provider (isang provider lang), at bawat koneksyon/key --**Color-Coded Proxy Badges**— Visual indicator: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, palaging ipinapakita ang IP --**OAuth Token Exchange Through Proxy**— Ang daloy ng OAuth ay dumadaan din sa proxy, na nilulutas ang `unsupported_country_region_territory` --**Mga Pagsusuri sa Koneksyon sa pamamagitan ng Proxy**— Ginagamit ng mga pagsubok sa koneksyon ang naka-configure na proxy (wala nang direktang bypass) --**SOCKS5 Support**— Buong SOCKS5 proxy support para sa papalabas na pagruruta --**TLS Fingerprint Spoofing**— Tulad ng browser na TLS fingerprint sa pamamagitan ng `wreq-js` para i-bypass ang bot detection --**🔏 CLI Fingerprint Matching**— Muling inaayos ang mga header at body field upang tumugma sa mga native na CLI binary signature, na lubhang binabawasan ang panganib sa pag-flag ng account. Ang proxy IP ay napanatili — nakakakuha ka ng parehong stealth**at**IP masking nang sabay
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. - -🆓 4. "Gusto kong gumamit ng AI para sa coding ngunit wala akong pera" +**How OmniRoute solves it:** -Hindi lahat ay maaaring magbayad ng $20–200/buwan para sa mga subscription sa AI. Ang mga mag-aaral, mga dev mula sa mga umuusbong na bansa, mga hobbyist, at mga freelancer ay nangangailangan ng access sa mga de-kalidad na modelo sa zero cost. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Paano ito niresolba ng OmniRoute:** + --**Libreng Tier Provider Built-in**— Native na suporta para sa 100% libreng provider: Qoder (5 unlimited na mga modelo sa pamamagitan ng OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwender3-wenlash3-coplus qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID nang libre), Gemini CLI (180K token/buwan libre) --**Ollama Cloud**— Cloud-hosted Ollama models sa `api.ollama.com` na may libreng "Light usage" tier; gumamit ng `ollamacloud/` prefix --**Free-Only Combos**— Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/buwan na walang downtime --**NVIDIA NIM Free Access**— ~40 RPM dev-forever na libreng access sa 70+ na modelo sa build.nvidia.com (paglilipat mula sa mga credit patungo sa mga purong limitasyon sa rate) --**Cost Optimized Strategy**— Istratehiya sa pagruruta na awtomatikong pinipili ang pinakamurang available na provider +
+🆓 4. "I want to use AI for coding but I have no money" - -🔒 5. "Kailangan kong protektahan ang aking AI gateway mula sa hindi awtorisadong pag-access" +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -Kapag inilantad ang isang gateway ng AI sa network (LAN, VPS, Docker), maaaring kumonsumo ng mga token/quota ng developer ang sinumang may address. Kung walang proteksyon, ang mga API ay mahina sa maling paggamit, agarang pag-iniksyon, at pang-aabuso. +**How OmniRoute solves it:** -**Paano ito niresolba ng OmniRoute:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**API Key Management**— Pagbuo, pag-ikot, at saklaw ng bawat provider na may nakalaang pahina ng `/dashboard/api-manager` --**Mga Pahintulot sa Antas ng Modelo**— Limitahan ang mga API key sa mga partikular na modelo (`openai/*`, mga pattern ng wildcard), na may toggle na Allow All/Restrict --**API Endpoint Protection**— Mangangailangan ng key para sa `/v1/models` at i-block ang mga partikular na provider mula sa listahan --**Auth Guard + CSRF Protection**— Lahat ng mga ruta ng dashboard ay protektado ng `withAuth` middleware + CSRF token --**Rate Limiter**— Per-IP rate na naglilimita sa mga na-configure na window --**IP Filtering**— Allowlist/blocklist para sa access control --**Prompt Injection Guard**— Sanitization laban sa malisyosong prompt pattern --**AES-256-GCM Encryption**— Ang mga kredensyal ay naka-encrypt sa pahinga
+ - -🛑 6. "Bumaba ang provider ko at nawala ang coding flow ko" +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -Ang mga tagapagbigay ng AI ay maaaring maging hindi matatag, magbalik ng 5xx na mga error, o maabot ang mga pansamantalang limitasyon sa rate. Kung ang isang dev ay nakadepende sa isang provider, maaantala sila. Kung walang mga circuit breaker, ang mga paulit-ulit na pagsubok ay maaaring mag-crash sa application. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Paano ito niresolba ng OmniRoute:** +**How OmniRoute solves it:** --**Circuit Breaker per-model**— Auto-open/close na may mga na-configure na threshold at cooldown (Closed/Open/Half-Open), scoped per-model para maiwasan ang mga cascading block --**Exponential Backoff**— Progressive retry delays --**Anti-Thundering Herd**— Mutex + semaphore na proteksyon laban sa kasabay na muling pagsubok na mga bagyo --**Combo Fallback Chains**— Kung nabigo ang pangunahing provider, awtomatikong mahuhulog sa chain nang walang interbensyon --**Combo Circuit Breaker**— Awtomatikong idi-disable ang mga nabigong provider sa loob ng combo chain --**Health Dashboard**— Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest - -🔧 7. "Nakakapagod at paulit-ulit ang pag-configure sa bawat AI tool" + -Gumagamit ang mga developer ng Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Ang bawat tool ay nangangailangan ng ibang config (API endpoint, key, model). Ang muling pag-configure kapag lumipat ng mga provider o modelo ay isang pag-aaksaya ng oras. +
+🛑 6. "My provider went down and I lost my coding flow" -**Paano ito niresolba ng OmniRoute:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**CLI Tools Dashboard**— Nakatuon na page na may isang-click na setup para sa Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline --**GitHub Copilot Config Generator**— Bumubuo ng `chatLanguageModels.json` para sa VS Code na may maramihang pagpili ng modelo --**Onboarding Wizard**— May gabay na 4-step na pag-setup para sa mga unang beses na user --**Isang endpoint, lahat ng modelo**— I-configure ang `http://localhost:20128/v1` nang isang beses, i-access ang 60+ provider
+**How OmniRoute solves it:** - -🔑 8. "Impiyerno ang pamamahala sa mga token ng OAuth mula sa maraming provider" +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot — lahat ay gumagamit ng OAuth 2.0 na may mga mag-e-expire na token. Kailangang muling mag-authenticate ang mga developer, harapin ang `client_secret is missing`, `redirect_uri_mismatch`, at mga pagkabigo sa mga malalayong server. Ang OAuth sa LAN/VPS ay partikular na may problema. + -**Paano ito niresolba ng OmniRoute:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Auto Token Refresh**— Ang mga token ng OAuth ay nagre-refresh sa background bago mag-expire --**OAuth 2.0 (PKCE) Built-in**— Awtomatikong daloy para sa Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder --**Multi-Account OAuth**— Maramihang account bawat provider sa pamamagitan ng pagkuha ng token ng JWT/ID --**OAuth LAN/Remote Fix**— Pribadong IP detection para sa `redirect_uri` + manual URL mode para sa mga malalayong server --**OAuth Behind Nginx**— Gumagamit ng `window.location.origin` para sa reverse proxy compatibility --**Remote OAuth Guide**— Step-by-step na gabay para sa mga kredensyal ng Google Cloud sa VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. - -📊 9. "Hindi ko alam kung magkano ang ginagastos ko o kung saan" +**How OmniRoute solves it:** -Gumagamit ang mga developer ng maraming bayad na provider ngunit walang pinag-isang pagtingin sa paggastos. Ang bawat provider ay may sariling dashboard ng pagsingil, ngunit walang pinagsama-samang view. Ang mga hindi inaasahang gastos ay maaaring magtambak. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Paano ito niresolba ng OmniRoute:** + --**Cost Analytics Dashboard**— Per-token cost tracking at pamamahala ng badyet bawat provider --**Mga Limitasyon sa Badyet bawat Tier**— Paggastos ng kisame sa bawat tier na nagti-trigger ng awtomatikong fallback --**Per-Model Pricing Configuration**— Nako-configure na mga presyo bawat modelo --**Mga Istatistika ng Paggamit Bawat API Key**— Bilang ng kahilingan at timestamp na huling ginamit bawat key --**Analytics Dashboard**— Mga stat card, chart ng paggamit ng modelo, talahanayan ng provider na may mga rate ng tagumpay at latency +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" - -🐛 10. "Hindi ko ma-diagnose ang mga error at problema sa AI calls" +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Kapag nabigo ang isang tawag, hindi alam ng dev kung ito ay isang limitasyon sa rate, nag-expire na token, maling format, o error sa provider. Mga fragment na log sa iba't ibang terminal. Kung walang pagmamasid, ang pag-debug ay trial-and-error. +**How OmniRoute solves it:** -**Paano ito niresolba ng OmniRoute:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Unified Logs Dashboard**— 4 na tab: Request Logs, Proxy Logs, Audit Logs, Console --**Console Log Viewer**— Real-time na terminal-style viewer na may color-coded level, auto-scroll, paghahanap, filter --**SQLite Proxy Logs**— Mga paulit-ulit na log na nakaligtas sa pag-restart ng server --**Translator Playground**— 4 na mode ng pag-debug: Playground (pagsasalin ng format), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) --**Request Telemetry**— p50/p95/p99 latency + X-Request-Id tracing --**File-Based Logging na may Rotation**— Ang mga log ng app ay umiikot ayon sa laki, araw ng pagpapanatili, at bilang ng archive; ang mga call log artifact ay umiikot ayon sa mga araw ng pagpapanatili at bilang ng file --**System Info Report**— Ang `npm run system-info` ay bumubuo ng `system-info.txt` kasama ang iyong buong environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Ilakip ito kapag nag-uulat ng mga isyu para sa instant triage.
+ - -🏗️ 11. "Ang pag-deploy at pagpapanatili ng gateway ay kumplikado" +
+📊 9. "I don't know how much I'm spending or where" -Ang pag-install, pag-configure, at pagpapanatili ng AI proxy sa iba't ibang kapaligiran (lokal, VPS, Docker, cloud) ay labor-intensive. Ang mga problema tulad ng mga hardcoded na path, `EACCES` sa mga direktoryo, port conflict, at cross-platform build ay nagdaragdag ng friction. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Paano ito niresolba ng OmniRoute:** +**How OmniRoute solves it:** --**npm global install**— `npm install -g omniroute && omniroute` — tapos na --**Docker Multi-Platform**— AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) --**Docker Compose Profiles**— `base` (walang CLI tool) at `cli` (na may Claude Code, Codex, OpenClaw) --**Electron Desktop App**— Native app para sa Windows/macOS/Linux na may system tray, auto-start, offline mode --**Split-Port Mode**— API at Dashboard sa magkahiwalay na port para sa mga advanced na sitwasyon (reverse proxy, container networking) --**Cloud Sync**— I-configure ang pag-synchronize sa mga device sa pamamagitan ng Cloudflare Workers --**DB Backup**— Awtomatikong pag-backup, pagpapanumbalik, pag-export at pag-import ng lahat ng mga setting, na may `DISABLE_SQLITE_AUTO_BACKUP` para sa mga external na pinamamahalaang backup
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency - -🌍 12. "Ang interface ay English-only at ang aking team ay hindi nagsasalita ng English" + -Ang mga koponan sa mga bansang hindi nagsasalita ng Ingles, lalo na sa Latin America, Asia, at Europe, ay nakikipagpunyagi sa mga interface na Ingles lamang. Binabawasan ng mga hadlang sa wika ang pag-aampon at pinapataas ang mga error sa pagsasaayos. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Paano ito niresolba ng OmniRoute:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Dashboard i18n — 30 Wika**— Lahat ng 500+ key na isinalin kabilang ang Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, --**RTL Support**— Kanan-pakaliwa na suporta para sa Arabic at Hebrew --**Multi-Language READMEs**— 30 kumpletong pagsasalin ng dokumentasyon --**Language Selector**— Globe icon sa header para sa real-time na paglipat
+**How OmniRoute solves it:** - -🔄 13. "Kailangan ko ng higit pa sa chat — kailangan ko ng mga embed, larawan, audio" +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -Ang AI ay hindi lamang pagkumpleto ng chat. Kailangan ng mga dev na bumuo ng mga larawan, mag-transcribe ng audio, gumawa ng mga pag-embed para sa RAG, mag-rerank ng mga dokumento, at katamtamang nilalaman. Ang bawat API ay may iba't ibang endpoint at format. + -**Paano ito niresolba ng OmniRoute:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Embeddings**— `/v1/embeddings` na may 6 na provider at 9+ na modelo --**Pagbuo ng Larawan**— `/v1/mga larawan/mga henerasyon` na may 10 provider at 20+ modelo (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Text-to-Video**— `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) at SD WebUI --**Text-to-Music**— `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) --**Audio Transcription**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Text-to-Speech**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + mga kasalukuyang provider --**Moderations**— `/v1/moderations` — Mga pagsusuri sa kaligtasan ng content --**Reranking**— `/v1/rerank` — Muling pagraranggo ng kaugnayan ng dokumento --**Responses API**— Buong `/v1/responses` na suporta para sa Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. - -🧪 14. "Wala akong paraan para subukan at paghambingin ang kalidad sa mga modelo" +**How OmniRoute solves it:** -Gustong malaman ng mga developer kung aling modelo ang pinakamainam para sa kanilang kaso ng paggamit — code, pagsasalin, pangangatwiran — ngunit mabagal ang paghahambing nang manu-mano. Walang pinagsamang mga tool sa eval ang umiiral. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**Paano ito niresolba ng OmniRoute:** + --**LLM Evaluations**— Golden set testing na may 10 pre-loaded na case na sumasaklaw sa mga pagbati, matematika, heograpiya, pagbuo ng code, pagsunod sa JSON, pagsasalin, markdown, pagtanggi sa kaligtasan --**4 na Istratehiya sa Pagtutugma**— `eksakto`, `naglalaman`, `regex`, `custom` (JS function) --**Translator Playground Test Bench**— Batch testing na may maraming input at inaasahang output, cross-provider na paghahambing --**Chat Tester**— Buong round-trip na may visual response rendering --**Live Monitor**— Real-time na stream ng lahat ng kahilingang dumadaloy sa proxy +
+🌍 12. "The interface is English-only and my team doesn't speak English" - -📈 15. "Kailangan kong sukatin nang hindi nawawala ang performance" +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -Habang lumalaki ang dami ng kahilingan, nang walang pag-cache sa parehong mga tanong ay bumubuo ng mga dobleng gastos. Nang walang idempotency, humihiling ang duplicate sa pagpoproseso ng basura. Dapat igalang ang mga limitasyon sa rate ng bawat provider. +**How OmniRoute solves it:** -**Paano ito niresolba ng OmniRoute:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Semantic Cache**— Ang two-tier na cache (pirma + semantiko) ay binabawasan ang gastos at latency --**Request Idempotency**— 5s deduplication window para sa magkaparehong mga kahilingan --**Pagtukoy sa Limitasyon ng Rate**— RPM ng bawat provider, min na gap, at max na kasabay na pagsubaybay --**Editable Rate Limits**— Configurable defaults in Settings → Resilience with persistence --**API Key Validation Cache**— 3-tier na cache para sa performance ng produksyon --**Health Dashboard na may Telemetry**— p50/p95/p99 latency, cache stats, uptime
+ - -🤖 16. "Gusto kong kontrolin ang gawi ng modelo sa buong mundo" +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Mga developer na gusto ang lahat ng tugon sa isang partikular na wika, na may partikular na tono, o gustong limitahan ang mga token ng pangangatwiran. Ang pag-configure nito sa bawat tool/kahilingan ay hindi praktikal. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Paano ito niresolba ng OmniRoute:** +**How OmniRoute solves it:** --**System Prompt Injection**— Inilapat ang pandaigdigang prompt sa lahat ng kahilingan --**Thinking Budget Validation**— Reasoning token allocation control bawat kahilingan (passthrough, auto, custom, adaptive) --**9 Mga Istratehiya sa Pagruruta**— Mga pandaigdigang diskarte na tumutukoy kung paano ipinamamahagi ang mga kahilingan --**Wildcard Router**— dynamic na ruta ng mga pattern ng `provider/*` sa anumang provider --**Combo Enable/Disable Toggle**— I-toggle ang mga combo nang direkta mula sa dashboard --**Toggle ng Provider**— I-enable/i-disable ang lahat ng koneksyon para sa isang provider sa isang click --**Mga Naka-block na Provider**— Ibukod ang mga partikular na provider mula sa listahan ng `/v1/models`
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex - -🧰 17. "Kailangan ko ng mga tool ng MCP bilang mga first-class na kakayahan ng produkto" + -Maraming AI gateway ang naglalantad sa MCP bilang isang nakatagong detalye ng pagpapatupad. Ang mga koponan ay nangangailangan ng isang nakikita, napapamahalaang layer ng operasyon. +
+🧪 14. "I have no way to test and compare quality across models" -**Paano ito niresolba ng OmniRoute:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- Lumilitaw ang MCP sa dashboard navigation at tab ng endpoint protocol -- Nakatuon na pahina ng pamamahala ng MCP na may proseso, mga tool, saklaw, at pag-audit -- Built-in na quick-start para sa `omniroute --mcp` at onboarding ng kliyente
+**How OmniRoute solves it:** - -🧠 18. "Kailangan ko ng A2A orchestration na may sync + stream task paths" +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Ang mga daloy ng trabaho ng ahente ay nangangailangan ng parehong direktang tugon at matagal na naka-stream na pagpapatupad na may kontrol sa lifecycle. + -**Paano ito niresolba ng OmniRoute:** +
+📈 15. "I need to scale without losing performance" -- A2A JSON-RPC endpoint (`POST /a2a`) na may `message/send` at `message/stream` -- SSE streaming na may terminal state propagation -- Mga task lifecycle API para sa `tasks/get` at `tasks/cancel`
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. - -🛰️ 19. "Kailangan ko ng totoong kalusugan ng proseso ng MCP, hindi nahulaan ang status" +**How OmniRoute solves it:** -Kailangang malaman ng mga operational team kung talagang buhay ang MCP, hindi lang kung maaabot ang isang API. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**Paano ito niresolba ng OmniRoute:** + -- Runtime heartbeat file na may PID, timestamp, transport, bilang ng tool, at mode ng saklaw -- MCP status API na pinagsasama ang tibok ng puso + kamakailang aktibidad -- Mga UI status card para sa pagiging bago ng proseso/uptime/heartbeat +
+🤖 16. "I want to control model behavior globally" - -📋 20. "Kailangan ko ng auditable MCP tool execution" +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Kapag ang mga tool ay nag-mutate ng config o nag-trigger ng mga pagkilos ng ops, ang mga team ay nangangailangan ng forensic traceability. +**How OmniRoute solves it:** -**Paano ito niresolba ng OmniRoute:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- SQLite-backed audit logging para sa mga tawag sa tool ng MCP -- Mga filter ayon sa tool, tagumpay/kabiguan, API key, at pagination -- Dashboard audit table + stats endpoints para sa automation
+ - -🔐 21. "Kailangan ko ng mga saklaw na pahintulot ng MCP sa bawat pagsasama" +
+🧰 17. "I need MCP tools as first-class product capabilities" -Ang iba't ibang mga kliyente ay dapat magkaroon ng hindi gaanong pribilehiyong pag-access sa mga kategorya ng tool. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Paano ito niresolba ng OmniRoute:** +**How OmniRoute solves it:** -- 10 butil na saklaw ng MCP para sa kontroladong pag-access ng tool -- Pagpapatupad ng saklaw at kakayahang makita sa UI ng pamamahala ng MCP -- Ligtas na default na postura para sa operational tooling
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding - -⚙️ 22. "Kailangan ko ng operational controls nang walang redeploying" + -Ang mga koponan ay nangangailangan ng mabilis na mga pagbabago sa runtime sa panahon ng mga insidente o mga kaganapan sa gastos. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**Paano ito niresolba ng OmniRoute:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Lumipat ng combo activation nang direkta mula sa MCP dashboard -- Ilapat ang mga profile ng katatagan mula sa paunang natukoy na mga pack ng patakaran -- I-reset ang estado ng circuit breaker mula sa parehong panel ng mga operasyon
+**How OmniRoute solves it:** - -🔄 23. "Kailangan ko ng live A2A task lifecycle visibility at cancellation" +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Kung walang lifecycle visibility, ang mga insidente ng gawain ay nagiging mahirap subukan. + -**Paano ito niresolba ng OmniRoute:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Listahan ng gawain/pag-filter ayon sa estado/kasanayan sa pagination -- Mag-drill-down sa metadata ng gawain, mga kaganapan, at mga artifact -- Endpoint ng pagkansela ng gawain at pagkilos ng UI na may kumpirmasyon
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. - -🌊 24. "Kailangan ko ng active stream metrics para sa A2A load" +**How OmniRoute solves it:** -Ang mga stream ng workflow ay nangangailangan ng operational insight sa concurrency at live na koneksyon. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**Paano ito niresolba ng OmniRoute:** + -- Mga aktibong stream counter na isinama sa A2A status -- Mga bilang ng huling timestamp ng gawain at bawat estado -- A2A dashboard card para sa real-time na pagsubaybay sa ops +
+📋 20. "I need auditable MCP tool execution" - -🪪 25. "Kailangan ko ng karaniwang pagtuklas ng ahente para sa mga kliyente" +When tools mutate config or trigger ops actions, teams need forensic traceability. -Ang mga panlabas na kliyente at orkestra ay nangangailangan ng metadata na nababasa ng makina para sa onboarding. +**How OmniRoute solves it:** -**Paano ito niresolba ng OmniRoute:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- Na-expose ang Agent Card sa `/.well-known/agent.json` -- Mga kakayahan at kasanayan na ipinapakita sa management UI -- Kasama sa A2A status API ang metadata ng pagtuklas para sa automation
+ - -🧭 26. "Kailangan ko ng protocol discoverability sa UX ng produkto" +
+🔐 21. "I need scoped MCP permissions per integration" -Kung hindi matuklasan ng mga user ang mga surface ng protocol, bumababa ang kalidad ng pag-aampon at suporta. +Different clients should have least-privilege access to tool categories. -**Paano ito niresolba ng OmniRoute:** +**How OmniRoute solves it:** -- Pinagsama-samang**Mga Endpoint**na pahina na may mga tab para sa Proxy, MCP, A2A, at API Endpoints -- Mga toggle ng katayuan ng inline na serbisyo (Online/Offline) para sa MCP at A2A -- Mga link mula sa pangkalahatang-ideya hanggang sa nakalaang mga tab ng pamamahala
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling - -🧪 27. "Kailangan ko ng end-to-end protocol validation sa mga totoong kliyente" + -Ang mga kunwaring pagsubok ay hindi sapat upang patunayan ang pagiging tugma ng protocol bago ilabas. +
+⚙️ 22. "I need operational controls without redeploying" -**Paano ito niresolba ng OmniRoute:** +Teams need quick runtime changes during incidents or cost events. -- E2E suite na nagbo-boot ng app at gumagamit ng totoong MCP SDK client transport -- Mga pagsubok sa A2A client para sa pagtuklas, pagpapadala, pag-stream, pagkuha, at pagkansela ng mga daloy -- Cross-check assertion laban sa MCP audit at A2A tasks API
+**How OmniRoute solves it:** - -📡 28. "Kailangan ko ng pinag-isang observability sa lahat ng interface" +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -Ang paghahati ng observability sa pamamagitan ng protocol ay lumilikha ng mga blind spot at mas mahabang MTTR. + -**Paano ito niresolba ng OmniRoute:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Pinag-isang mga dashboard/log/analytics sa isang produkto -- Health + audit + humiling ng telemetry sa mga layer ng OpenAI, MCP, at A2A -- Operational APIs for status and automation
+Without lifecycle visibility, task incidents become hard to triage. - -💼 29. "Kailangan ko ng isang runtime para sa proxy + tools + orchestration ng ahente" +**How OmniRoute solves it:** -Ang pagpapatakbo ng maraming magkakahiwalay na serbisyo ay nagpapataas ng gastos sa pagpapatakbo at mga mode ng pagkabigo. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**Paano ito niresolba ng OmniRoute:** + -- OpenAI-compatible na proxy, MCP server, at A2A server sa isang stack -- Nakabahaging auth, resilience, data store, at observability -- Pare-parehong modelo ng patakaran sa lahat ng surface ng pakikipag-ugnayan +
+🌊 24. "I need active stream metrics for A2A load" - -🚀 30. "Kailangan kong magpadala ng mga ahenteng daloy ng trabaho nang walang glue-code sprawl" +Streaming workflows require operational insight into concurrency and live connections. -Nawawalan ng bilis ang mga koponan kapag nagtatahi ng maraming ad-hoc na serbisyo at script. +**How OmniRoute solves it:** -**Paano ito niresolba ng OmniRoute:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Pinag-isang endpoint na diskarte para sa mga kliyente at ahente -- Mga built-in na UI sa pamamahala ng protocol at mga daanan sa pagpapatunay ng usok -- Mga pundasyong handa sa produksyon (seguridad, pag-log, katatagan, backup)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Playbook A: I-maximize ang bayad na subscription + murang backup**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -607,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Playbook B: Zero-cost coding stack**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Playbook C: 24/7 always-on fallback chain**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -630,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Playbook D: Ahente ops sa MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> I-setup ang AI coding sa ilang minuto sa**$0/buwan**. Ikonekta ang mga libreng account na ito at gamitin ang built-in na**Free Stack**combo. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Hakbang | Aksyon | Na-unlock ang Mga Provider | -| ---- | ---------------------------------------------------- | -------------------------------------------------------------------- | -| 1 | Ikonekta ang**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**walang limitasyon**| -| 2 | Ikonekta ang**Qoder**(Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... —**unlimited**| -| 3 | Ikonekta ang**Qwen**(Device Code) | qwen3-coder-plus, qwen3-coder-flash... —**walang limitasyon**| -| 4 | Ikonekta ang**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180K/mo libre**| -| 5 | `/dashboard/combos` →**Libreng Stack ($0)**template | Awtomatikong round-robin lahat ng libreng provider | +| Step | Action | Providers Unlocked | +| ---- | -------------------------------------------------- | ------------------------------------------------------------------ | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Ituro ang anumang IDE/CLI sa:**`http://localhost:20128/v1` · API Key: `any-string` · Tapos na. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Opsyonal na karagdagang coverage (libre din):**Groq API key (30 RPM libre), NVIDIA NIM (40 RPM libre, 70+ modelo), Cerebras (1M tok/araw), LongCat API key (50M token/araw!), Cloudflare Workers AI (10K Neurons/araw, 50+ modelo).## Mabilis na Simula +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Mabilis na Simula ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **mga user ng pnpm:**Patakbuhin ang `pnpm approve-builds -g` pagkatapos i-install upang paganahin ang mga native na build script na kinakailangan ng `better-sqlite3` at `@swc/core`: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > > ```bash > pnpm install -g omniroute -> pnpm approve-builds -g # Piliin ang lahat ng package → aprubahan +> pnpm approve-builds -g # Select all packages → approve > omniroute > ``` -Magbubukas ang dashboard sa `http://localhost:20128` at ang API base URL ay `http://localhost:20128/v1`. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Utos | Paglalarawan | -| ----------------------- | -------------------------------------------------------------------- | -| `omniroute` | Simulan ang server (`PORT=20128`, API at dashboard sa parehong port) | -| `omniroute --port 3000` | Itakda ang canonical/API port sa 3000 | -| `omniroute --mcp` | Simulan ang MCP server (stdio transport) | -| `omniroute --no-open` | Huwag awtomatikong buksan ang browser | -| `omniroute --help` | Ipakita ang tulong | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Opsyonal na split-port mode:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -Para sa karamihan ng mga deployment, kailangan mo lang: +For most deployments, you only need: -| Variable | Default | Layunin | -| ------------------------- | ----------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------ | -| `REQUEST_TIMEOUT_MS` | `600000` | Nakabahaging baseline para sa upstream na pagkuha, mga nakatagong Undici timeout, mga kahilingan sa TLS fingerprint, at kahilingan sa API bridge/proxy timeout | -| `STREAM_IDLE_TIMEOUT_MS` | nagmamana ng `REQUEST_TIMEOUT_MS` | Pinakamataas na agwat sa pagitan ng mga streaming chunks bago i-abort ng OmniRoute ang SSE stream | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -Pinapanatili ang backward compatibility: ang umiiral na `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, at iba pang per-layer timeout vars ay gumagana pa rin at na-override ang nakabahaging baseline. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Available ang mga advanced na override kung kailangan mo ng mas pinong kontrol:| Variable | Default | Layunin | +Advanced overrides are available if you need finer control: + +| Variable | Default | Purpose | | ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | nagmamana ng `REQUEST_TIMEOUT_MS` | Kabuuang pag-timeout ng upstream na kahilingan na ginagamit ng pangunahing signal ng pag-abort ng pagkuha | -| `FETCH_HEADERS_TIMEOUT_MS` | nagmamana ng `FETCH_TIMEOUT_MS` | Undici time limit para sa pagtanggap ng upstream response header | -| `FETCH_BODY_TIMEOUT_MS` | nagmamana ng `FETCH_TIMEOUT_MS` | Undici time limit sa pagitan ng upstream body chunks (`0` idi-disable ito) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | -| `FETCH_KEEPLIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | -| `TLS_CLIENT_TIMEOUT_MS` | nagmamana ng `FETCH_TIMEOUT_MS` | Timeout para sa mga kahilingan sa fingerprint ng TLS na ginawa sa pamamagitan ng `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | nagmamana ng `REQUEST_TIMEOUT_MS` o `30000` | Timeout para sa `/v1` proxy forwarding mula sa API port papunta sa dashboard port | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Timeout ng papasok na kahilingan sa API bridge server | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Papasok na header timeout sa API bridge server | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout sa API bridge server | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout sa API bridge server (nadi-disable ito ng `0`) | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -Kung nagpapatakbo ka ng OmniRoute sa likod ng Nginx, Caddy, Cloudflare, o isa pang reverse proxy, tiyaking ang proxy -ang mga timeout ay mas mataas din kaysa sa iyong OmniRoute stream/fetch timeout.### 2) Connect providers and create your API key +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. -1. Buksan ang Dashboard → `Mga Provider` at ikonekta ang hindi bababa sa isang provider (OAuth o API key). -2. Buksan ang Dashboard → `Endpoints` at gumawa ng API key. -3. (Opsyonal) Buksan ang Dashboard → `Combos` at itakda ang iyong fallback chain.### 3) Point your coding tool to OmniRoute +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Gumagana sa Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, at OpenAI-compatible SDK.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (para sa mga pagpapatakbong hinihimok ng tool):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -Pagkatapos ay ikonekta ang iyong MCP client sa `stdio` at mga tool sa pagsubok tulad ng: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (para sa mga daloy ng trabaho ng ahente-sa-agent):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -759,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Ang suite na ito ay nagpapatunay ng totoong MCP at A2A na mga daloy ng kliyente laban sa isang tumatakbong app.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -767,13 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` - +
Void Linux (`xbps-src` template) -Para sa mga gumagamit ng Void Linux, maaari kang bumuo ng native package gamit ang `xbps-src`. I-save ang block na ito bilang `srcpkgs/omniroute/template`:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -785,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -793,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -869,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -880,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -Available ang OmniRoute bilang pampublikong larawan ng Docker sa [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Mabilis na pagtakbo:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -890,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**Na may environment file:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Gumagamit ng Docker Compose:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Kasama na ngayon sa suporta sa dashboard para sa mga deployment ng Docker ang isang isang-click na**Cloudflare Quick Tunnel**sa `Dashboard → Endpoints`. Ang unang paganahin ang mga pag-download ng `cloudflared` lamang kapag kinakailangan, magsisimula ng pansamantalang tunnel sa iyong kasalukuyang `/v1` endpoint, at ipinapakita ang nabuong `https://*.trycloudflare.com/v1` URL nang direkta sa ibaba ng iyong normal na pampublikong URL. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Mga Tala: +Notes: -- Ang mga URL ng Quick Tunnel ay pansamantala at nagbabago pagkatapos ng bawat pag-restart. -- Ang Mga Mabilisang Tunnel ay hindi na-auto-restore pagkatapos ng OmniRoute o pag-restart ng container. Muling paganahin ang mga ito mula sa dashboard kung kinakailangan. -- Kasalukuyang sinusuportahan ng pinamamahalaang pag-install ang Linux, macOS, at Windows sa `x64` / `arm64`. -- Default ang Managed Quick Tunnels sa HTTP/2 transport para maiwasan ang maingay na QUIC UDP buffer na babala sa mga napipigilan na kapaligiran ng container. Itakda ang `CLOUDFLARED_PROTOCOL=quic` o `auto` kung gusto mo ng ibang sasakyan. -- Pinagsasama ng mga larawan ng Docker ang mga ugat ng CA ng system at ipinapasa ang mga ito sa pinamamahalaang `cloudflared`, na nag-iwas sa mga pagkabigo ng tiwala ng TLS kapag nag-bootstrap ang tunnel sa loob ng container. -- Tumatakbo ang SQLite sa WAL mode. Ang `docker stop` ay dapat pahintulutang matapos upang masuri ng OmniRoute ang mga pinakabagong pagbabago pabalik sa `storage.sqlite`. -- Nagtakda na ng 40s stop na palugit ang mga naka-bundle na Compose file. Kung direkta mong patakbuhin ang larawan, panatilihing `--stop-timeout 40` (o katulad nito) upang hindi maputol ang pag-shutdown ng mga manual stop. -- Itakda ang `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` kung gusto mong gumamit ang OmniRoute ng umiiral nang binary sa halip na mag-download ng isa. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Paggamit ng Docker Compose with Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -Ang OmniRoute ay maaaring ligtas na mailantad gamit ang awtomatikong SSL provisioning ni Caddy. Tiyaking tumuturo ang DNS A record ng iyong domain sa IP ng iyong server.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` +| Image | Tag | Size | Description | +| ------------------------ | -------- | ------ | --------------------- | +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | -| Larawan | Tag | Sukat | Paglalarawan | -| ------------------------- | -------- | ------ | ---------------------- | -| `diegosouzapw/omniroute` | `pinakabago` | ~250MB | Pinakabagong stable na release | -| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Kasalukuyang bersyon |--- +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**BAGO!**Available na ngayon ang OmniRoute bilang**katutubong desktop application**para sa Windows, macOS, at Linux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Patakbuhin ang OmniRoute bilang isang standalone na desktop app — walang terminal, walang browser, walang internet na kailangan para sa mga lokal na modelo. Kasama sa Electron-based na app ang: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Native Window**— Nakatuon na window ng app na may integration ng system tray -- 🔄**Auto-Start**— Ilunsad ang OmniRoute sa system login -- 🔔**Mga Katutubong Notification**— Makakuha ng mga alerto para sa pagkaubos ng quota o mga isyu sa provider -- ⚡**One-Click Install**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Offline Mode**— Gumagana nang ganap offline sa naka-bundle na server### Mabilis na Simula +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Mabilis na Simula ```bash # Development mode @@ -979,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -Kapag pinaliit, nakatira ang OmniRoute sa iyong system tray na may mabilis na pagkilos: +When minimized, OmniRoute lives in your system tray with quick actions: -- Buksan ang dashboard -- Baguhin ang port ng server -- Ihinto ang aplikasyon +- Open dashboard +- Change server port +- Quit application -📖 Buong dokumentasyon: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Tier | Provider | Gastos | I-reset ang Quota | Pinakamahusay Para sa | -| ------------------- | --------------------------------- | ------------------------------- | -------------------- | ------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/buwan | 5h + lingguhan | Naka-subscribe na | -| | Codex (Plus/Pro) | $20-200/buwan | 5h + lingguhan | Mga user ng OpenAI | -| | Gemini CLI | **LIBRE** | 180K/buwan + 1K/araw | Lahat! | -| | GitHub Copilot | $10-19/buwan | Buwanang | Mga user ng GitHub | -| **🔑 API KEY** | NVIDIA NIM | **LIBRE**(dev forever) | ~40 RPM | 70+ bukas na mga modelo | -| | Cerebras | **LIBRE**(1M tok/araw) | 60K TPM / 30 RPM | Pinakamabilis sa mundo | -| | Groq | **LIBRE**(30 RPM) | 14.4K RPD | Napakabilis na Llama/Gemma | -| | DeepSeek V3.2 | $0.27/$1.10 bawat 1M | Wala | Pinakamahusay na presyo/kalidad na pangangatwiran | -| | xAI Grok-4 Mabilis | **$0.20/$0.50 bawat 1M**🆕 | Wala | Pinakamabilis + tool na pagtawag, ultralow | -| | xAI Grok-4 (standard) | $0.20/$1.50 bawat 1M 🆕 | Wala | Nangangatuwirang punong barko mula sa xAI | -| | Mistral | Libreng pagsubok + bayad | Limitado ang rate | European AI | -| | OpenRouter | Pay-per-use | Wala | 100+ modelo aggr. | -| **💰 MURA** | GLM-5 (sa pamamagitan ng Z.AI) 🆕 | $0.5/1M | Araw-araw 10AM | 128K output, pinakabagong flagship | -| | GLM-4.7 | $0.6/1M | Araw-araw 10AM | Backup ng badyet | -| | MiniMax M2.5 🆕 | $0.3/1M input | 5 oras na rolling | Pangangatwiran + mga ahenteng gawain | -| | MiniMax M2.1 | $0.2/1M | 5 oras na rolling | Pinaka murang opsyon | -| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | Wala | Direktang Moonshot API access | -| | Kimi K2 | $9/buwan flat | 10M token/buwan | Nahuhulaang gastos | -| **🆓 LIBRE** | Qoder | **$0** | Walang limitasyong | 5 mga modelong walang limitasyon | -| | Qwen | **$0** | Walang limitasyong | 4 na modelong walang limitasyon | -| | Kiro | **$0** | Walang limitasyong | Claude Sonnet/Haiku (AWS Builder) | -| | LongCat Flash-Lite 🆕 | **$0**(50M tok/araw 🔥) | 1 RPS | Pinakamalaking libreng quota sa Earth | -| | Mga polinasyon AI 🆕 | **$0**(walang key na kailangan) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | -| | Cloudflare Workers AI 🆕 | **$0**(10K Neuron/araw) | ~150 resp/araw | 50+ modelo, pandaigdigang gilid | -| | Scaleway AI 🆕 | **$0**(1M token sa kabuuan) | Limitado ang rate | EU/GDPR, Qwen3 235B, Llama 70B | > 🆕**Idinagdag ang mga bagong modelo (Mar 2026):**Grok-4 Fast family sa $0.20/$0.50/M (na-benchmark sa 1143ms — 30% mas mabilis kaysa Gemini 2.5 Flash), GLM-5 sa pamamagitan ng Z.AI na may 128K na output, MiniMax M2.5 na pangangatwiran, na-update ng Kim2 na direktang pagpepresyo V3. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 $0 Combo Stack — Ang Kumpletong Libreng Setup:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Walang halaga. Hindi kailanman tumitigil sa pag-coding.**I-configure ito bilang isang combo ng OmniRoute at lahat ng fallback ay awtomatikong nangyayari — walang manu-manong paglipat kailanman.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Lahat ng mga modelo sa ibaba ay**100% libre na walang kinakailangang credit card**. Ang mga awtomatikong ruta ng OmniRoute sa pagitan ng mga ito kapag naubos ang isang quota — pagsamahin silang lahat para sa hindi nababasag na $0 combo.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Modelo | Prefix | Limitahan | Hangganan ng Rate | -| ------------------- | ------ | ------------- | ---------------------- | -| `claude-sonnet-4.5` | `kr/` |**Walang limitasyon**| Walang iniulat na pang-araw-araw na cap | -| `claude-haiku-4.5` | `kr/` |**Walang limitasyon**| Walang iniulat na pang-araw-araw na cap | -| `claude-opus-4.6` | `kr/` |**Walang limitasyon**| Pinakabagong Opus sa pamamagitan ng Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) -| Modelo | Prefix | Limitahan | Hangganan ng Rate | -| ------------------- | ------ | ------------- | --------------- | -| `kimi-k2-thinking` | `kung/` |**Walang limitasyon**| Walang naiulat na cap | -| `qwen3-coder-plus` | `kung/` |**Walang limitasyon**| Walang naiulat na cap | -| `deepseek-r1` | `kung/` |**Walang limitasyon**| Walang naiulat na cap | -| `minimax-m2.1` | `kung/` |**Walang limitasyon**| Walang naiulat na cap | -| `kimi-k2` | `kung/` |**Walang limitasyon**| Walang naiulat na cap | +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | --------------------- | +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -> Inirerekomendang paraan ng koneksyon:**Personal Access Token + `qodercli`**. Ang browser OAuth ay -> pang-eksperimento at hindi pinagana bilang default maliban kung ang mga variable ng kapaligiran ng `QODER_OAUTH_*` ay na-configure.### 🟡 QWEN MODELS (Device Code Auth) +### 🟢 QODER MODELS (Free PAT via qodercli) -| Modelo | Prefix | Limitahan | Hangganan ng Rate | +| Model | Prefix | Limit | Rate Limit | +| ------------------ | ------ | ------------- | --------------- | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | + +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. + +### 🟡 QWEN MODELS (Device Code Auth) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | ------------------- | -| `qwen3-coder-plus` | `qw/` |**Walang limitasyon**| Walang naiulat na cap | -| `qwen3-coder-flash` | `qw/` |**Walang limitasyon**| Walang naiulat na cap | -| `qwen3-coder-next` | `qw/` |**Walang limitasyon**| Walang naiulat na cap | -| `modelo ng paningin` | `qw/` |**Walang limitasyon**| Multimodal (mga larawan) |### 🟣 GEMINI CLI (Google OAuth) +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Modelo | Prefix | Limitahan | Hangganan ng Rate | -| ------------------------- | ------ | --------------------------- | ------------- | -| `gemini-3-flash-preview` | `gc/` |**180K tok/buwan**+ 1K/araw | Buwanang pag-reset | -| `gemini-2.5-pro` | `gc/` | 180K/buwan (nakabahaging pool) | Mataas na kalidad |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +### 🟣 GEMINI CLI (Google OAuth) -| Tier | Pang-araw-araw na Limitasyon | Hangganan ng Rate | Mga Tala | -| ---------- | ------------ | ----------- | ------------------------------------------------------------------- | -| Libre (Dev) | Walang token cap |**~40 RPM**| 70+ modelo; paglipat sa purong mga limitasyon sa rate sa kalagitnaan ng 2025 | +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -Mga sikat na libreng modelo: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-deepseek`, `deepseek-instruct`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) -| Tier | Pang-araw-araw na Limitasyon | Hangganan ng Rate | Mga Tala | -| ---- | ----------------- | ---------------- | ------------------------------------------ | -| Libre |**1M token/araw**| 60K TPM / 30 RPM | Pinakamabilis na LLM inference sa mundo; nire-reset araw-araw | +| Tier | Daily Limit | Rate Limit | Notes | +| ---------- | ------------ | ----------- | ------------------------------------------------------ | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -Available nang libre: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -| Tier | Pang-araw-araw na Limitasyon | Hangganan ng Rate | Mga Tala | -| ---- | ------------- | ---------------- | ---------------------------------------- | -| Libre |**14.4K RPD**| 30 RPM bawat modelo | Walang credit card; 429 sa limitasyon, hindi sinisingil | +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -Available nang libre: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -| Modelo | Prefix | Araw-araw na Libreng Quota | Mga Tala | -| ----------------------------- | ------ | ----------------- | ------------------------ | -| `LongCat-Flash-Lite` | `lc/` |**50M token**💥 | Pinakamalaking libreng quota kailanman | -| `LongCat-Flash-Chat` | `lc/` | 500K token | Multi-turn chat | -| `LongCat-Flash-Thinking` | `lc/` | 500K token | Pangangatwiran / CoT | -| `LongCat-Flash-Thinking-2601` | `lc/` | 500K token | Ene 2026 na bersyon | -| `LongCat-Flash-Omni-2603` | `lc/` | 500K token | Multimodal | +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -> 100% libre habang nasa pampublikong beta. Mag-sign up sa [longcat.chat](https://longcat.chat) gamit ang email o telepono. Nire-reset araw-araw 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +### 🔴 GROQ (Free API Key — console.groq.com) -| Modelo | Prefix | Hangganan ng Rate | Provider sa Likod | -| ---------- | ------ | ---------- | ------------------- | -| `openai` | `pol/` | 1 req/15s | GPT-5 | -| `claude` | `pol/` | 1 req/15s | Anthropic Claude | -| `gemini` | `pol/` | 1 req/15s | Google Gemini | -| `malalim' | `pol/` | 1 req/15s | DeepSeek V3 | -| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | -| `mistral` | `pol/` | 1 req/15s | Mistral AI | +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ------------- | ---------------- | ----------------------------------------- | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -> ✨**Zero friction:**Walang pag-signup, walang API key. Idagdag ang provider ng Pollinations na may walang laman na key field at agad itong gagana.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -| Tier | Pang-araw-araw na Neuron | Katumbas na Paggamit | Mga Tala | -| ---- | ------------- | --------------------------------------- | ------------------------ | -| Libre |**10,000**| ~150 LLM resp / 500s audio / 15K na pag-embed | Global edge, 50+ na modelo | +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 -Mga sikat na libreng modelo: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (libreng audio!), `@cf/qwen/qwen2.5-coder`15b +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | -> Nangangailangan ng API Token + Account ID mula sa [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID sa mga setting ng provider.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. -| Tier | Libreng Quota | Lokasyon | Mga Tala | +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | +| ---------- | ------ | ---------- | ------------------ | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | + +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. + +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 + +| Tier | Daily Neurons | Equivalent Usage | Notes | +| ---- | ------------- | --------------------------------------- | ----------------------- | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | + +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` + +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. + +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | | ---- | ------------- | ------------ | ----------------------------------- | -| Libre |**1M token**| 🇫🇷 Paris, EU | Walang kinakailangang credit card sa loob ng mga limitasyon | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | -Available nang libre: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` -> Sumusunod sa EU/GDPR. Kunin ang API key sa [console.scaleway.com](https://console.scaleway.com). +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). ->**💡 Ang Ultimate Free Stack (11 Provider, $0 Forever):** +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M token/araw 🔥 -> Mga polinasyon (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — walang susi na kailangan -> Qwen (qw/) → qwen3-coder models UNLIMITED -> Gemini (gemini/) → Gemini 2.5 Flash — libre ang 1,500 req/araw -> Cloudflare AI (cf/) → 50+ na modelo — 10K Neurons/araw -> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M libreng token (EU) -> Groq (groq/) → Llama/Gemma — 14.4K req/araw na napakabilis -> NVIDIA NIM (nvidia/) → 70+ bukas na modelo — 40 RPM magpakailanman -> Cerebras (cerebras/) → Llama/Qwen pinakamabilis sa mundo — 1M tok/araw -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> I-transcribe ang anumang audio/video para sa**$0**— Nangunguna ang Deepgram na may libreng $200, AssemblyAI $50 fallback, Groq Whisper bilang walang limitasyong backup na pang-emergency. +## 🎙️ Free Transcription Combo -| Provider | Libreng Credits | Pinakamahusay na Modelo | Hangganan ng Rate | -| ----------------- | ----------------------- | ------------------------------------------ | ---------------------------- | -| 🟢**Deepgram**|**$200 libre**(pag-signup) | `nova-3` — pinakamahusay na katumpakan, 30+ wika | Walang limitasyon sa RPM sa mga libreng kredito | -| 🔵**AssemblyAI**|**$50 libre**(pag-signup) | `universal-3-pro` — mga kabanata, damdamin, PII | Walang limitasyon sa RPM sa mga libreng kredito | -| 🔴**Groq**|**Libre magpakailanman**| `whisper-large-v3` — OpenAI Whisper | 30 RPM (limitado ang rate) | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. -**Iminungkahing combo sa `/dashboard/combos`:**``` +| Provider | Free Credits | Best Model | Rate Limit | +| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | + +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Pagkatapos ay sa `/dashboard/media` →**Transcription**tab: mag-upload ng anumang audio o video file → piliin ang iyong combo endpoint → kumuha ng transkripsyon sa mga sinusuportahang format.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -Ang OmniRoute v2.0 ay binuo bilang isang operating platform, hindi lamang isang relay proxy.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Tampok | Ano ang Ginagawa Nito | -| ---------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Grok-4 Mabilis na Pamilya** | mga modelo ng xAI sa $0.20/$0.50/M — na-benchmark na 1143ms (30% mas mabilis kaysa sa Gemini 2.5 Flash) | -| 🧠**GLM-5 sa pamamagitan ng Z.AI** | 128K na konteksto ng output, $0.5/1M — pinakabagong flagship mula sa pamilyang GLM | -| 🔮**MiniMax M2.5** | Pangangatwiran + mga ahenteng gawain sa $0.30/1M — makabuluhang pag-upgrade mula sa M2.1 | -| 🎯**toolCalling Flag bawat Model** | Per-modelo na `toolCalling: true/false` sa registry — Nilaktawan ng AutoCombo ang mga modelong hindi may kakayahan sa tool | -| 🌍**Multilingual Intent Detection** | Mga keyword ng PT/ZH/ES/AR sa AutoCombo scoring — mas mahusay na pagpili ng modelo para sa nilalamang hindi Ingles | -| 📊**Mga Fallback na Dahil sa Benchmark** | Real p95 latency mula sa mga live na kahilingan feed combo scoring — natututo ang AutoCombo mula sa aktwal na data | -| 🔁**Humiling ng Deduplication** | Content-hash based na dedup window — multi-agent safe, pinipigilan ang mga duplicate na singil | -| 🔌**Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — magdagdag ng custom na routing logic bilang mga plugin | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Tampok | Ano ang Ginagawa Nito | -| -------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Modelo Playground** | Dashboard page upang direktang subukan ang anumang modelo — provider/modelo/endpoint selector, Monaco Editor, streaming, abort, timing | -| 🔏**CLI Fingerprint Matching** | Pag-order ng header/body ng bawat provider upang tumugma sa mga native na lagda ng CLI — i-toggle ang bawat provider sa Mga Setting > Seguridad.**Napanatili ang iyong proxy IP** | -| 🤝**Suporta sa ACP (Agent Client Protocol)** | Pagtuklas ng ahente ng CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 pa), process spawner, `/api/acp/agents` endpoint | -| 🤖**Dashboard ng Mga Ahente ng ACP** | Debug › Pahina ng mga ahente — grid ng 14 na ahente na may katayuan sa pag-install, bersyon, custom na form ng ahente para sa anumang CLI tool. Ang**OpenCode**na mga user ay nakakakuha ng button na "I-download ang opencode.json" na awtomatikong bumubuo ng isang ready-to-use na config kasama ang lahat ng available na modelo. | -| 🔧**Custom Model `apiFormat` Routing** | Ang mga custom na modelo na may `apiFormat: "mga tugon"` ay tama na ngayon ang ruta sa tagasalin ng Responses API | -| 🏢**Paghihiwalay ng Codex Workspace** | Maramihang mga workspace ng Codex bawat email — Tamang pinaghihiwalay ng OAuth ang mga koneksyon ayon sa workspace ID | -| 🔄**Electron Auto-Update** | Sinusuri ng desktop app ang mga update + auto-install sa pag-restart | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Tampok | Ano ang Ginagawa Nito | -| ------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------- | -| 🔧**MCP Server (25 tool)** | Mga tool ng IDE/agent sa pamamagitan ng 3 transport: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 na tool sa kasanayan | -| 🤝**A2A Server (JSON-RPC + SSE)** | Pagpapatupad ng gawain ng ahente-sa-agent na may pag-sync at mga daloy ng streaming | -| 🧭**Consolidated Endpoints Page** | Pahina ng pamamahala sa tab na may mga tab na Endpoint Proxy, MCP, A2A, at Mga Endpoint ng API | -| 🎚️**Service Enable/Disable Toggles** | ON/OFF switch para sa MCP at A2A na may mga setting ng persistence (default: OFF) | -| 🛰️**MCP Runtime Heartbeat** | Katayuan ng totoong proseso (pid, uptime, edad ng tibok ng puso, transportasyon, mode ng saklaw) | -| 📋**MCP Audit Trail** | Na-filter na mga log ng pag-audit na may tagumpay/kabiguan at pangunahing pagpapatungkol | -| 🔐**Pagpapatupad ng Saklaw ng MCP** | 10 butil na saklaw na pahintulot para sa kontroladong pag-access ng tool | -| 📡**A2A Task Lifecycle Management** | Ilista/i-filter ang mga gawain, siyasatin ang mga kaganapan/artifact, kanselahin ang pagpapatakbo ng mga gawain | -| 📋**Pagtuklas ng Ahente Card** | `/.well-known/agent.json` para sa auto-discovery ng kliyente | -| 🧪**Protocol E2E Test Harness** | Ang totoong MCP SDK + A2A client ay dumadaloy sa `test:protocols:e2e` | -| ⚙️**Mga Operational Control** | Lumipat ng combo, ilapat ang mga profile ng resilience, i-reset ang mga breaker mula sa isang control surface | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Tampok | Ano ang Ginagawa Nito | -| ---------------------------------------- | ---------------------------------------------------------------------------------------------- | ----------------------- | -| 🎯**Smart 4-Tier Fallback** | Auto-ruta: Subscription → API Key → Mura → Libre | -| 📊**Real-Time Quota Tracking** | Live na bilang ng token + reset countdown bawat provider | -| 🔄**Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Mga tugon na may mga conversion na ligtas sa schema | -| 👥**Suporta sa Multi-Account** | Maramihang account sa bawat provider na may matalinong pagpili | -| 🔄**Auto Token Refresh** | Awtomatikong nagre-refresh ang mga token ng OAuth sa muling pagsubok | -| 🎨**Mga Custom na Combos** | 9 na diskarte sa pagbabalanse + fallback chain control | -| 🌐**Wildcard Router** | `provider/*` dynamic na pagruruta | -| 🧠**Mga Kontrol sa Badyet sa Pag-iisip** | Passthrough, auto, custom, at adaptive na mga limitasyon sa pangangatwiran | -| 🔀**Mga Alyas ng Modelo** | Built-in + custom na model aliasing at kaligtasan sa paglilipat | -| ⚡**Pagbaba ng Background** | Iruta ang mga gawain sa background na mababa ang priyoridad sa mas murang mga modelo | -| 🧪**Task-Aware Smart Routing** | Awtomatikong piliin ang modelo ayon sa uri ng nilalaman (coding/vision/analysis/summarization) | -| 🔄**Mga Workflow ng Ahente ng A2A** | Deterministic FSM orchestrator para sa stateful multi-step agent executions | -| 🔀**Adaptive Routing** | Pag-override ng dinamikong diskarte batay sa dami ng token at agarang pagiging kumplikado | -| 🎲**Diversity ng Provider** | Shannon entropy scoring balancing auto-combo traffic distribution | -| 💬**System Prompt Injection** | Patuloy na inilalapat ang mga pandaigdigang kontrol sa gawi | -| 📄**Pagkatugma sa API ng Mga Tugon** | Buong suporta sa `/v1/responses` para sa Codex at mga advanced na ahenteng daloy ng trabaho | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Tampok | Ano ang Ginagawa Nito | -| ------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**Pagbuo ng Larawan** | `/v1/images/generations` na may cloud at mga lokal na backend | -| 📐**Mga Pag-embed** | `/v1/embeddings` para sa paghahanap at mga pipeline ng RAG | -| 🎤**Audio Transcription** | `/v1/audio/transcriptions` — 7 provider (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, suporta sa MP4/MP3/WAV | -| 🔊**Text-to-Speech** | `/v1/audio/speech` — 10 provider (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) na may mga tamang mensahe ng error | -| 🎬**Pagbuo ng Video** | `/v1/video/generations` (ComfyUI + SD WebUI workflows) | -| 🎵**Pagbuo ng Musika** | `/v1/music/generations` (ComfyUI workflows) | -| 🛡️**Mga Pag-moderate** | `/v1/moderations` mga pagsusuri sa kaligtasan | -| 🔀**Reranking** | `/v1/rerank` para sa pagmamarka ng kaugnayan | -| 🔍**Paghahanap sa Web**🆕 | `/v1/search` — 5 provider (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ libre/buwan, auto-failover, cache | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Tampok | Ano ang Ginagawa Nito | -| --------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**Mga Circuit Breaker** | Bawat modelong biyahe/pagbawi na may mga kontrol sa threshold | -| 🎯**Mga Modelo ng Endpoint-Aware** | Idineklara ng mga custom na modelo ang mga sinusuportahang endpoint + format ng API | -| 🛡️**Anti-Thundering Herd** | Mga proteksyon ng Mutex + semaphore sa muling pagsubok/pag-rate ng mga kaganapan | -| 🧠**Semantic + Signature Cache** | Pagbabawas ng gastos/latency na may dalawang layer ng cache | -| ⚡**Humiling ng Idempotency** | Duplicate na window ng proteksyon | -| 🔒**TLS Fingerprint Spoofing** | TLS fingerprint na parang browser —**binabawasan ang pag-detect ng bot at pag-flag ng account** | -| 🔏**CLI Fingerprint Matching** | Tumutugma sa mga native na lagda ng kahilingan sa CLI —**binabawasan ang panganib sa pagbabawal habang pinapanatili ang proxy IP** | -| 🌐**Pag-filter ng IP** | Allowlist/blocklist control para sa mga nakalantad na deployment | -| 📊**Mga Nae-edit na Limitasyon sa Rate** | Nako-configure ang mga limitasyon sa antas ng global/provider na may pagtitiyaga | -| 📉**Graceful Degradation** | Mga fallback ng multi-layer na kakayahan na nagpoprotekta sa mga pangunahing pagpapatakbo ng gateway | -| 📜**Config Audit Trail** | Diff-based na pagsubaybay sa pagbabago na pumipigil sa pag-anod ng pagpapatakbo gamit ang mga simpleng rollback | -| ⏳**Provider Health Sync** | Proactive token expiration monitoring na nagti-trigger ng mga alerto bago ang mga pagkabigo ng pahintulot | -| 🚪**Awtomatikong I-disable ang Mga Banned Account** | Ang operational circuit breaker na nagse-sealing ng permanenteng na-block na mga token account ay awtomatikong | -| 🔑**API Key Management + Scoping** | Secure na pagpapalabas/pag-ikot ng susi at mga kontrol ng modelo/tagapagbigay | -| 👁️**Scoped API Key Reveal**🆕 | Mag-opt-in sa pagbawi ng mga API key sa pamamagitan ng `ALLOW_API_KEY_REVEAL` | -| 🛡️**Protektado `/models`** | Opsyonal na auth gating at pagtatago ng provider para sa catalog ng modelo | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Tampok | Ano ang Ginagawa Nito | -| ------------------------------------------ | -------------------------------------------------------------------- | ---------------------------- | -| 📝**Kahilingan + Proxy Logging** | Buong kahilingan/tugon at proxy logging | -| 📉**Mga Naka-stream na Detalyadong Log**🆕 | Inaayos muli ang mga SSE payload stream nang malinis sa UI | -| 📋**Unified Logs Dashboard** | Kahilingan, proxy, audit, at console view sa isang page | -| 🔍**Humiling ng Telemetry** | p50/p95/p99 latency at kahilingan sa pagsubaybay | -| 🏥**Dashboard ng Kalusugan** | Uptime, breaker states, lockouts, cache stats | -| 💰**Pagsubaybay sa Gastos** | Mga kontrol sa badyet at visibility ng pagpepresyo sa bawat modelo | -| 📈**Mga Visualization ng Analytics** | Mga insight sa paggamit ng modelo/provider at mga view ng trend | -| 🧪**Brangkas ng Pagsusuri** | Golden set testing na may mga configurable na diskarte sa pagtutugma | -| 📡**Live Diagnostics**🆕 | Semantic cache bypass para sa tumpak na combo live na pagsubok | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Tampok | Ano ang Ginagawa Nito | -| ---------------------------------- | -------------------------------------------------------------------------------------------------------------- | --------------------- | -| 🌐**I-deploy Kahit Saan** | Localhost, VPS, Docker, Cloud environment | -| 🚇**Cloudflare Tunnel**🆕 | Isang-click na Quick Tunnel integration mula sa dashboard | -| 🔑**API Key Model Filtering** | Na-filter na tugon ng katutubong /v1/models sa pamamagitan ng mga itinalagang tungkulin sa konteksto ng Bearer | -| ⚡**Smart Cache Bypass** | Configurable TTL heuristics at sapilitang refetch na mga kontrol | -| 🔄**Backup/Restore** | Mga daloy ng pag-export/pag-import at pagbawi ng kalamidad | -| 🧙**Onboarding Wizard** | First-run guided setup | -| 🔧**CLI Tools Dashboard** | Isang-click na setup para sa mga sikat na coding tool | -| 🎮**Modelo Playground** | Subukan ang anumang provider/modelo/endpoint mula sa dashboard | -| 🔏**CLI Fingerprint Toggle** | Pagtutugma ng fingerprint ng bawat provider sa Mga Setting > Seguridad | -| 🌐**i18n (30 wika)** | Buong dashboard + suporta sa wika ng mga doc na may saklaw ng RTL | -| 🧹**I-clear ang Lahat ng Modelo** | Isang-click na pag-clear ng listahan ng modelo sa mga detalye ng provider | -| 👁️**Mga Kontrol sa Sidebar**🆕 | Itago ang mga bahagi at pagsasama mula sa Mga Setting ng Hitsura | -| 📋**Mga Template ng Isyu** | Standardized GitHub template para sa mga bug at feature | -| 📂**Custom na Direktoryo ng Data** | Pag-override ng `DATA_DIR` para sa lokasyon ng storage | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1292,78 +1464,99 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -Kapag nabigo ang quota, rate, o kalusugan, awtomatikong lilipat ang OmniRoute sa susunod na kandidato nang walang manu-manong paglipat.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- Ang MCP + A2A ay natutuklasan sa UI at mga doc (hindi nakatago) -- Inilalantad ng mga Protocol status API ang live na data ng pagpapatakbo (`/api/mcp/*`, `/api/a2a/*`) -- Kasama sa mga dashboard ang mga aksyon para sa day-2 ops (combo toggle, breaker reset, pagkansela ng gawain)#### Translator + validation workflow +#### Protocol management that is visible and operable -Kasama sa lugar ng Tagasalin ang: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Playground**: humiling ng mga pagsusuri sa pagbabago -**Chat Tester**: buong kahilingan/tugon round-trip -**Test Bench**: maraming kaso sa isang pagtakbo -**Live Monitor**: real-time na view ng trapiko +#### Translator + validation workflow -Dagdag pa ang pagpapatunay ng protocol sa mga tunay na kliyente sa pamamagitan ng `npm run test:protocols:e2e`. +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Tool reference, IDE config, at mga halimbawa ng client +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A Server README](src/lib/a2a/README.md)**— Mga Kasanayan, JSON-RPC na pamamaraan, streaming, at lifecycle ng gawain## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -Ang OmniRoute ay may kasamang built-in na balangkas ng pagsusuri upang subukan ang kalidad ng pagtugon ng LLM laban sa isang ginintuang hanay. I-access ito sa pamamagitan ng**Analytics → Evals**sa dashboard.### Built-in Golden Set +## 🧪 Evaluations (Evals) -Ang pre-loaded na "OmniRoute Golden Set" ay naglalaman ng mga test case para sa: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Pagbati, matematika, heograpiya, pagbuo ng code -- Pagsunod sa format ng JSON, pagsasalin, pagbuo ng markdown -- Pagtanggi sa kaligtasan (nakapipinsalang nilalaman), pagbibilang, lohika ng boolean### Evaluation Strategies +### Built-in Golden Set -| Diskarte | Paglalarawan | Halimbawa | -| ------------ | ------------------------------------------------------------ | -------------------------------- | --- | -| `eksakto` | Dapat na eksaktong tumugma ang output | `"4"` | -| `naglalaman` | Ang output ay dapat maglaman ng substring (case-insensitive) | `"Paris"` | -| `regex` | Ang output ay dapat tumugma sa regex pattern | `"1.*2.*3"` | -| `pasadya` | Ang custom na JS function ay nagbabalik ng true/false | `(output) => output.length > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) - +
🧩 MCP Setup (Model Context Protocol) -Simulan ang MCP transport sa stdio mode:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Inirerekomendang daloy ng pagpapatunay: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Ikonekta ang iyong MCP client sa stdio. -2. Patakbuhin ang `omniroute_get_health`. -3. Patakbuhin ang `omniroute_list_combos`. -4. Buksan ang `/dashboard/mcp` para kumpirmahin ang tibok ng puso, aktibidad, at pag-audit. - -Mga kapaki-pakinabang na API para sa automation: +Useful APIs for automation: - `GET /api/mcp/status` - `GET /api/mcp/tools` - `GET /api/mcp/audit` -- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats` - -🤝 A2A Setup (Agent2Agent) + -Tuklasin ang ahente:```bash +
+🤝 A2A Setup (Agent2Agent) + +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Magpadala ng gawain:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` - -Pamahalaan ang lifecycle: +Manage lifecycle: - `GET /api/a2a/status` - `GET /api/a2a/tasks` @@ -1372,23 +1565,31 @@ Pamahalaan ang lifecycle: Operational UI: -- `/dashboard/a2a` para sa gawain/estado/pagmamasid sa stream at mga pagkilos sa usok
+- `/dashboard/a2a` for task/state/stream observability and smoke actions - + + +
🧪 End-to-end protocol validation -I-validate ang parehong protocol sa mga totoong kliyente:```bash +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Ito ay nagpapatunay: +This verifies: -- MCP SDK client kumonekta/listahan/tawag +- MCP SDK client connect/list/call - A2A discovery/send/stream/get/cancel -- Cross-check ang data sa MCP audit at A2A task management API
+- Cross-check data in MCP audit and A2A task management APIs - -💳 Subscription Provider### Claude Code (Pro/Max) + + +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1401,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Pro Tip:**Gamitin ang Opus para sa mga kumplikadong gawain, Soneto para sa bilis. Sinusubaybayan ng OmniRoute ang quota bawat modelo!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1415,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Ang bawat Codex account ay mayroon na ngayong mga toggle ng patakaran sa `Dashboard -> Mga Provider`: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (ON/OFF): ipatupad ang 5-hour window threshold policy. -- `Lingguhan` (ON/OFF): ipatupad ang lingguhang patakaran sa threshold ng window. -- Threshold na gawi: kapag ang isang naka-enable na window ay umabot sa >=90% na paggamit, ang account na iyon ay nilalaktawan. -- Pag-uugali ng pag-ikot: Awtomatikong mga ruta ang OmniRoute patungo sa susunod na karapat-dapat na Codex account. -- I-reset ang gawi: kapag lumipas ang oras ng provider na `resetAt`, awtomatikong magiging kwalipikadong muli ang account. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Mga sitwasyon: +Scenarios: -- `5h ON` + `Weekly ON`: ang account ay nilalaktawan kapag ang alinman sa window ay umabot sa threshold. -- `5h OFF` + `Weekly ON`: lingguhang paggamit lang ang makaka-block sa account. -- `5h ON` + `Lingguhang OFF`: 5 oras na paggamit lang ang makaka-block sa account. -- Lumipas ang `resetAt`: awtomatikong pumapasok ang account sa pag-ikot (walang manual na muling paganahin).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1440,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Pinakamahusay na Halaga:**Malaking libreng tier! Gamitin ito bago ang mga bayad na tier.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1455,71 +1662,91 @@ Models:
- -🔑 API Key Provider### NVIDIA NIM (FREE developer access — 70+ models) +
+🔑 API Key Providers -1. Mag-sign up: [build.nvidia.com](https://build.nvidia.com) -2. Kumuha ng libreng API key (1000 inference credits kasama) -3. Dashboard → Magdagdag ng Provider → NVIDIA NIM: +### NVIDIA NIM (FREE developer access — 70+ models) + +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: - API Key: `nvapi-your-key` -**Mga Modelo:**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, at 50+ pa +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -**Pro Tip:**OpenAI-compatible na API — gumagana nang walang putol sa pagsasalin ng format ng OmniRoute!### DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -1. Mag-sign up: [platform.deepseek.com](https://platform.deepseek.com) -2. Kunin ang API key -3. Dashboard → Magdagdag ng Provider → DeepSeek +### DeepSeek -**Mga Modelo:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -1. Mag-sign up: [console.groq.com](https://console.groq.com) -2. Kunin ang API key (kasama ang libreng tier) -3. Dashboard → Magdagdag ng Provider → Groq +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Mga Modelo:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +### Groq (Free Tier Available!) -**Pro Tip:**Napakabilis na hinuha — pinakamahusay para sa real-time na coding!### OpenRouter (100+ Models) +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -1. Mag-sign up: [openrouter.ai](https://openrouter.ai) -2. Kunin ang API key -3. Dashboard → Magdagdag ng Provider → OpenRouter +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Mga Modelo:**I-access ang 100+ na modelo mula sa lahat ng pangunahing provider sa pamamagitan ng iisang API key. +**Pro Tip:** Ultra-fast inference — best for real-time coding! -**Gawi sa Dashboard:**Ang mga modelo ng OpenRouter ay pinamamahalaan mula sa**Mga Magagamit na Modelo**. Ang manu-manong pagdaragdag, pag-import, at pag-auto-sync ay lahat ay nag-a-update sa parehong listahan.
+### OpenRouter (100+ Models) - -💰 Cheap Provider (Backup)### GLM-4.7 (Daily reset, $0.6/1M) +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -1. Mag-sign up: [Zhipu AI](https://open.bigmodel.cn/) -2. Kumuha ng API key mula sa Coding Plan -3. Dashboard → Magdagdag ng API Key: +**Models:** Access 100+ models from all major providers through a single API key. + +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. + + + +
+💰 Cheap Providers (Backup) + +### GLM-4.7 (Daily reset, $0.6/1M) + +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: - Provider: `glm` - API Key: `your-key` -**Gamitin:**`glm/glm-4.7` +**Use:** `glm/glm-4.7` -**Pro Tip:**Nag-aalok ang Coding Plan ng 3× na quota sa 1/7 na halaga! I-reset araw-araw 10:00 AM.### MiniMax M2.1 (5h reset, $0.20/1M) +**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. -1. Mag-sign up: [MiniMax](https://www.minimax.io/) -2. Kunin ang API key -3. Dashboard → Magdagdag ng API Key +### MiniMax M2.1 (5h reset, $0.20/1M) -**Gamitin:**`minimax/MiniMax-M2.1` +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key -**Pro Tip:**Ang pinakamurang opsyon para sa mahabang konteksto (1M token)!### Kimi K2 ($9/month flat) +**Use:** `minimax/MiniMax-M2.1` -1. Mag-subscribe: [Moonshot AI](https://platform.moonshot.ai/) -2. Kunin ang API key -3. Dashboard → Magdagdag ng API Key +**Pro Tip:** Cheapest option for long context (1M tokens)! -**Gamitin:**`kimi/kimi-latest` +### Kimi K2 ($9/month flat) -**Pro Tip:**Nakapirming $9/buwan para sa 10M token = $0.90/1M epektibong gastos!
+1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key - -🆓 LIBRENG Provider (Emergency Backup)### Qoder (5 FREE models via OAuth) +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1560,8 +1787,10 @@ Models:
- -🎨 Lumikha ng Combos### Example 1: Maximize Subscription → Cheap Backup +
+🎨 Create Combos + +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1589,8 +1818,10 @@ Cost: $0 forever!
- -🔧 CLI Integration### Cursor IDE +
+🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1601,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Gamitin ang pahina ng**CLI Tools**sa dashboard para sa isang pag-click na configuration, o manu-manong i-edit ang `~/.claude/settings.json`.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1612,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Pagpipilian 1 — Dashboard (inirerekomenda):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Pagpipilian 2 — Manwal:**I-edit ang `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1629,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Tandaan:**Ang OpenClaw ay gumagana lamang sa lokal na OmniRoute. Gamitin ang `127.0.0.1` sa halip na `localhost` upang maiwasan ang mga isyu sa paglutas ng IPv6.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1643,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Hakbang 1:**Idagdag ang OmniRoute bilang custom na provider:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Hakbang 2:**Gumawa/mag-edit ng `opencode.json` sa ugat ng iyong proyekto:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1669,117 +1909,130 @@ opencode } } } -```` +``` -**Hakbang 3:**Piliin ang modelo sa OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Tip:**Magdagdag ng anumang modelong available sa iyong OmniRoute `/v1/models` endpoint sa seksyong `models`. Gamitin ang format na `provider/model-id` mula sa iyong OmniRoute dashboard.
+ --- ## Pag-troubleshoot - -I-click upang palawakin ang gabay sa pag-troubleshoot +
+Click to expand troubleshooting guide -**"Ang modelo ng wika ay hindi nagbigay ng mga mensahe"** +**"Language model did not provide messages"** -- Naubos na ang quota ng provider → Suriin ang tracker ng quota ng dashboard -- Solusyon: Gumamit ng combo fallback o lumipat sa mas murang tier +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**Paglilimita sa rate** +**Rate limiting** -- Out na ang quota ng subscription → Fallback sa GLM/MiniMax -- Magdagdag ng combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**Nag-expire ang token ng OAuth** +**OAuth token expired** -- Auto-refresh ng OmniRoute -- Kung magpapatuloy ang mga isyu: Dashboard → Provider → Muling kumonekta +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Mataas na gastos** +**High costs** -- Suriin ang mga istatistika ng paggamit sa Dashboard → Mga Gastos -- Ilipat ang pangunahing modelo sa GLM/MiniMax -- Gumamit ng libreng tier (Gemini CLI, Qoder) para sa mga hindi kritikal na gawain +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Mali ang mga dashboard/API port** +**Dashboard/API ports are wrong** -- Ang `PORT` ay ang canonical base port (at API port bilang default) -- Ino-override lang ng `API_PORT` ang OpenAI-compatible na API listener -- Ino-override lang ng `DASHBOARD_PORT` ang dashboard/Next.js listener -- Itakda ang `NEXT_PUBLIC_BASE_URL` sa iyong dashboard/pampublikong URL (para sa mga OAuth callback) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Mga error sa cloud sync** +**Cloud sync errors** -- I-verify ang `BASE_URL` na mga puntos sa iyong running instance -- I-verify ang `CLOUD_URL` na mga puntos sa iyong inaasahang cloud endpoint -- Panatilihing nakahanay ang mga value ng `NEXT_PUBLIC_*` sa mga value sa gilid ng server +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Hindi gumagana ang unang pag-login** +**First login not working** -- Lagyan ng check ang `INITIAL_PASSWORD` sa `.env` -- Kung hindi nakatakda, ang fallback na password ay `123456` +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Walang mga log ng kahilingan** +**No request logs** -- Ang mga artifact ng kahilingan ay isinulat sa `DATA_DIR/call_logs/` bilang isang JSON file bawat kahilingan -- I-enable ang pagkuha ng pipeline mula sa Dashboard → Logs → Request Logs kung kailangan mo ng detalyadong per-stage payloads -- Itakda ang `APP_LOG_TO_FILE=true` kung gusto mo rin ng application console logs sa `logs/application/app.log` -- Isaayos ang `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, at `CALL_LOG_MAX_ENTRIES` kung kinakailangan +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Ang pagsubok sa koneksyon ay nagpapakita ng "Di-wasto" para sa mga provider na katugma sa OpenAI** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Maraming provider ang hindi naglalantad ng `/models` endpoint -- Kasama sa OmniRoute v1.0.6+ ang fallback validation sa pamamagitan ng mga pagkumpleto ng chat -- Tiyaking may kasamang `/v1` na suffix ang base URL### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ Mahalaga para sa mga user na nagpapatakbo ng OmniRoute sa isang VPS, Docker, o anumang malayuang server**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -Ang**Antigravity**at**Gemini CLI**provider ay gumagamit ng**Google OAuth 2.0**. Kinakailangan ng Google ang `redirect_uri` sa daloy ng OAuth upang eksaktong tumugma sa isa sa mga paunang nakarehistrong URI sa Google Cloud Console ng app. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -Ang mga kredensyal ng OAuth na naka-bundle sa OmniRoute ay nakarehistro**para sa `localhost` lamang**. Kapag na-access mo ang OmniRoute sa isang malayuang server (hal. `https://omniroute.myserver.com`), tinatanggihan ng Google ang pagpapatotoo gamit ang:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Kailangan mong gumawa ng**OAuth 2.0 Client ID**sa Google Cloud Console gamit ang URI ng iyong server.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Buksan ang Google Cloud Console** +#### Step-by-step -Pumunta sa: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Gumawa ng bagong OAuth 2.0 Client ID** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- I-click ang**"+ Lumikha ng Mga Kredensyal"**→**"OAuth client ID"** -- Uri ng application:**"Web application"** -- Pangalan: anumang gusto mo (hal. `OmniRoute Remote`) +**2. Create a new OAuth 2.0 Client ID** -**3. Magdagdag ng mga Awtorisadong Redirect URI** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -Sa field na**"Authorized redirect URIs"**, idagdag ang:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Palitan ang `your-server.com` ng domain o IP ng iyong server (isama ang port kung kinakailangan, hal. `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. I-save at kopyahin ang mga kredensyal** +After creating, Google will show the **Client ID** and **Client Secret**. -Pagkatapos gumawa, ipapakita ng Google ang**Client ID**at**Client Secret**. +**5. Set environment variables** -**5. Magtakda ng mga variable ng kapaligiran** +In your `.env` (or Docker environment variables): -Sa iyong `.env` (o mga variable ng kapaligiran ng Docker):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1788,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. I-restart ang OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Subukang kumonekta muli** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Dashboard → Mga Provider → Antigravity (o Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Ire-redirect na ngayon ng Google nang tama sa `https://your-server.com/callback`.--- +--- #### Temporary workaround (without custom credentials) -Kung hindi mo gustong mag-set up ng sarili mong mga kredensyal sa ngayon, maaari mo pa ring gamitin ang**manual na daloy ng URL**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. Binubuksan ng OmniRoute ang URL ng awtorisasyon ng Google -2. Pagkatapos magpahintulot, sinusubukan ng Google na mag-redirect sa `localhost` (na nabigo sa remote server) -3.**Kopyahin ang buong URL**mula sa address bar ng iyong browser (kahit na hindi naglo-load ang page) -4. I-paste ang URL na iyon sa field na ipinapakita sa OmniRoute connection modal -5. I-click ang**"Kumonekta"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Gumagana ito dahil valid ang authorization code sa URL kahit na-load man ang redirect page.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. - -🇧🇷 Versão em Português#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Os provedores**Antigravity**at**Gemini CLI**gamit ang**Google OAuth 2.0**para sa autenticação. O Google exige que a `redirect_uri` use no fluxo OAuth seja**exatamente**das URIs pré-cadastradas no Google Cloud Console for aplicativo. +
+🇧🇷 Versão em Português -Bilang credenciais OAuth embutidas no OmniRoute estão cadastradas**apenas para sa `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (hal: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Você precisa criar um**OAuth 2.0 Client ID**walang Google Cloud Console com a URI do seu servidor.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Mag-access sa Google Cloud Console** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) **2. Crie um novo OAuth 2.0 Client ID** -- Clique em**"+ Lumikha ng Mga Kredensyal"**→**"OAuth client ID"** -- Tipo de aplicativo:**"Web application"** -- Pangalan: escolha qualquer nome (hal: `OmniRoute Remote`) +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Idagdag bilang Mga Awtorisadong URI sa Pag-redirect** +**3. Adicione as Authorized Redirect URIs** -Walang campo**"Mga Awtorisadong URI sa pag-redirect"**, idagdag:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Palitan ang `seu-servidor.com` pelo domínio ou IP do seu servidor (kasama ang porta se necessário, hal: `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. I-save at kopyahin bilang kredensyal** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Após criar, o Google mostrará o**Client ID**at**Client Secret**. +**5. Configure as variáveis de ambiente** -**5. I-configure bilang variáveis de ambiente** +No seu `.env` (ou nas variáveis de ambiente do Docker): -No seu `.env` (ou nas variáveis de ambiente do Docker):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1867,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reinicie o OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute - -```` +``` **7. Tente conectar novamente** -Dashboard → Mga Provider → Antigravity (ou Gemini CLI) → OAuth +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Agora o Google redirectionará corretamente para sa `https://seu-servidor.com/callback` at a autenticação funcionará.--- +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. + +--- #### Workaround temporário (sem configurar credenciais próprias) -Se não quiser criar credenciais próprias agora, may posibilidad na magamit o fluxo**manual de URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. OmniRoute abrirá a URL de authorização do Google -2. Após você authorizar, o Google tentará redirectionar para sa `localhost` (que falha no servidor remoto) -3.**Kopyahin ang isang URL completa**sa barra de endereço do seu browser (mesmo que a página não carregue) +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) 4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute -5. Clique em**"Kumonekta"** +5. Clique em **"Connect"** -> Ang workaround na ito ay gumagana sa pamamagitan ng código de authorização na URL ay maaaring mag-redirect sa iyong carregado ou não.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1905,64 +2171,73 @@ Se não quiser criar credenciais próprias agora, may posibilidad na magamit o f ## 🛠️ Tech Stack - -I-click upang palawakin ang mga detalye ng tech stack +
+Click to expand tech stack details --**Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ ay**hindi suportado**— hindi tugma ang `better-sqlite3` native binary) --**Language**: TypeScript 5.9 —**100% TypeScript**sa `src/` at `open-sse/` (zero `any` sa mga core module mula noong v2.0) --**Framework**: Next.js 16 + React 19 + Tailwind CSS 4 --**Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) --**Schemas**: Zod (MCP tool I/O validation, mga kontrata ng API) --**Mga Protocol**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Streaming**: Mga Kaganapang Ipinadala ng Server (SSE) --**Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization --**Pagsubok**: Node.js test runner + Vitest (900+ test kasama ang unit, integration, E2E) --**CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) --**Website**: [omniroute.online](https://omniroute.online) --**Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Dokumentasyon -| Dokumento | Paglalarawan | -| ---------------------------------------------- | ---------------------------------------------------- | -| [User Guide](docs/USER_GUIDE.md) | Mga provider, combo, CLI integration, deployment | -| [API Reference](docs/API_REFERENCE.md) | Lahat ng mga endpoint na may mga halimbawa | -| [MCP Server](open-sse/mcp-server/README.md) | 16 MCP tool, IDE configs, Python/TS/Go client | -| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, kasanayan, streaming, gawain mgmt | -| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode pack, self-healing | -| [Pag-troubleshoot](docs/TROUBLESHOOTING.md) | Mga karaniwang problema at solusyon | -| [Arkitektura](docs/ARCHITECTURE.md) | Arkitektura ng system at mga panloob | -| [Contributing](CONTRIBUTING.md) | Pag-setup at mga alituntunin ng pag-unlad | -| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 na detalye | -| [Patakaran sa Seguridad](SECURITY.md) | Pag-uulat ng kahinaan at mga kasanayan sa seguridad | -| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Kumpletong gabay: VM + nginx + Cloudflare setup | -| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour na may mga screenshot | -| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Mga hakbang sa pagpapatunay bago ang paglabas |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -Ang OmniRoute ay may**210+ feature na binalak**sa maraming yugto ng pag-unlad. Narito ang mga pangunahing lugar: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Kategorya | Mga Nakaplanong Tampok | Mga Highlight | -| ----------------------------- | ---------------- | ------------------------------------------------------------------------------------ | -| 🧠**Routing at Intelligence**| 25+ | Lowest-latency routing, tag-based na routing, quota preflight, P2C account selection | -| 🔒**Seguridad at Pagsunod**| 20+ | SSRF hardening, credential cloaking, rate-limit sa bawat endpoint, management key scoping | -| 📊**Pagmamasid**| 15+ | Pagsasama ng OpenTelemetry, real-time na pagsubaybay sa quota, pagsubaybay sa gastos bawat modelo | -| 🔄**Mga Pagsasama ng Provider**| 20+ | Dynamic na model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | -| ⚡**Pagganap**| 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | -| 🌐**Ecosystem**| 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**OpenCode Integration**— Suporta ng katutubong provider para sa OpenCode AI coding IDE -- 🔗**TRAE Integration**— Buong suporta para sa balangkas ng pag-develop ng TRAE AI -- 📦**Batch API**— Asynchronous na pagproseso ng batch para sa maramihang kahilingan -- 🎯**Tag-Based Routing**— Mga kahilingan sa ruta batay sa mga custom na tag at metadata -- 💰**Diskarte sa Pinakamababang Gastos**— Awtomatikong piliin ang pinakamurang available na provider +### 🔜 Coming Soon -> 📝 Available ang buong detalye ng feature sa [`docs/new-features/`](docs/new-features/) (217 detalyadong spec)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1970,18 +2245,20 @@ Ang OmniRoute ay may**210+ feature na binalak**sa maraming yugto ng pag-unlad. N ### How to Contribute -1. I-fork ang repository -2. Lumikha ng iyong sangay ng tampok (`git checkout -b feature/amazing-feature`) -3. I-commit ang iyong mga pagbabago (`git commit -m 'Add amazing feature'`) -4. Push sa branch (`git push origin feature/amazing-feature`) -5. Magbukas ng Pull Request +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Tingnan ang [CONTRIBUTING.md](CONTRIBUTING.md) para sa mga detalyadong alituntunin.### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1993,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Espesyal na pasasalamat kay**[9router](https://github.com/decolua/9router)**ni**[decolua](https://github.com/decolua)**— ang orihinal na proyektong nagbigay inspirasyon sa fork na ito. Bumubuo ang OmniRoute sa hindi kapani-paniwalang pundasyong iyon na may mga karagdagang feature, multi-modal na API, at buong TypeScript na muling pagsulat. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Espesyal na salamat sa**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— ang orihinal na pagpapatupad ng Go na nagbigay inspirasyon sa JavaScript port na ito.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Lisensya -MIT License - tingnan ang [LICENSE](LICENSE) para sa mga detalye.--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/phi/docs/ARCHITECTURE.md b/docs/i18n/phi/docs/ARCHITECTURE.md index f185322117..e38832b682 100644 --- a/docs/i18n/phi/docs/ARCHITECTURE.md +++ b/docs/i18n/phi/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Huling na-update: 2026-03-28_## Executive Summary -Ang OmniRoute ay isang lokal na AI routing gateway at dashboard na binuo sa Next.js. -Nagbibigay ito ng isang endpoint na katugma sa OpenAI (`/v1/*`) at niruruta ang trapiko sa maraming upstream provider na may pagsasalin, fallback, pag-refresh ng token, at pagsubaybay sa paggamit. -Mga pangunahing kakayahan: +_Last updated: 2026-03-28_ -- OpenAI-compatible na API surface para sa CLI/tools (28 provider) -- Kahilingan/tugon sa pagsasalin sa mga format ng provider -- Modelong combo fallback (multi-model sequence) -- Account-level fallback (multi-account bawat provider) -- Pamamahala ng koneksyon ng provider ng OAuth + API-key -- Pagbuo ng pag-embed sa pamamagitan ng `/v1/embeddings` (6 na provider, 9 na modelo) -- Pagbuo ng larawan sa pamamagitan ng `/v1/images/generations` (4 na provider, 9 na modelo) -- Isipin ang pag-parse ng tag (`...`) para sa mga modelo ng pangangatwiran -- Response sanitization para sa mahigpit na OpenAI SDK compatibility -- Pag-normalize ng tungkulin (developer→system, system→user) para sa cross-provider compatibility +## Executive Summary + +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. + +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility - Structured output conversion (json_schema → Gemini responseSchema) -- Lokal na pagtitiyaga para sa mga provider, key, alias, combo, setting, pagpepresyo -- Pagsubaybay sa paggamit/gastos at pag-log ng kahilingan -- Opsyonal na cloud sync para sa multi-device/state sync -- IP allowlist/blocklist para sa API access control -- Pag-iisip ng pamamahala sa badyet (passthrough/auto/custom/adaptive) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) - Global system prompt injection -- Pagsubaybay sa session at fingerprinting -- Paglilimita sa pinahusay na rate ng bawat account gamit ang mga profile na partikular sa provider -- Pattern ng circuit breaker para sa katatagan ng provider -- Proteksyon laban sa dumadagundong na kawan na may mutex locking -- Nakabatay sa lagda ang cache ng pag-deduplication ng kahilingan -- Layer ng domain: availability ng modelo, mga panuntunan sa gastos, patakaran sa fallback, patakaran sa lockout -- Pananatili ng estado ng domain (SQLite write-through cache para sa mga fallback, badyet, lockout, circuit breaker) -- Policy engine para sa sentralisadong pagsusuri ng kahilingan (lockout → budget → fallback) -- Humiling ng telemetry na may p50/p95/p99 latency aggregation -- Correlation ID (X-Request-Id) para sa end-to-end na pagsubaybay -- Pag-log sa audit ng pagsunod gamit ang opt-out sa bawat API key -- Eval framework para sa LLM quality assurance -- Resilience UI dashboard na may real-time na status ng circuit breaker -- Modular OAuth providers (12 indibidwal na module sa ilalim ng `src/lib/oauth/providers/`) +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) -Pangunahing modelo ng runtime: +Primary runtime model: -- Ang mga ruta ng Next.js app sa ilalim ng `src/app/api/*` ay nagpapatupad ng parehong dashboard API at compatibility API -- Isang nakabahaging SSE/routing core sa `src/sse/*` + `open-sse/*` ang humahawak sa pagpapatupad ng provider, pagsasalin, streaming, fallback, at paggamit## Scope and Boundaries +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Lokal na gateway runtime -- Mga API sa pamamahala ng dashboard -- Pagpapatunay ng provider at pag-refresh ng token -- Humiling ng pagsasalin at SSE streaming -- Lokal na estado + pagtitiyaga sa paggamit -- Opsyonal na cloud sync orchestration### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Pagpapatupad ng serbisyo sa cloud sa likod ng `NEXT_PUBLIC_CLOUD_URL` -- Provider SLA/control plane sa labas ng lokal na proseso -- Mga panlabas na CLI binary mismo (Claude CLI, Codex CLI, atbp.)## Dashboard Surface (Current) +### Out of Scope -Mga pangunahing pahina sa ilalim ng `src/app/(dashboard)/dashboard/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — mabilis na pagsisimula + pangkalahatang-ideya ng provider -- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint na mga tab -- `/dashboard/providers` — mga koneksyon at kredensyal ng provider -- `/dashboard/combos` — mga combo na diskarte, mga template, mga panuntunan sa pagruruta ng modelo -- `/dashboard/costs` — pagsasama-sama ng gastos at visibility ng pagpepresyo -- `/dashboard/analytics` — analytics ng paggamit at mga pagsusuri -- `/dashboard/limits` — mga kontrol sa quota/rate +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls - `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation -- `/dashboard/agents` — nakitang mga ahente ng ACP + pagpaparehistro ng custom na ahente -- `/dashboard/media` — larawan/video/palaruan ng musika -- `/dashboard/search-tools` — pagsubok at kasaysayan ng provider ng paghahanap -- `/dashboard/health` — uptime, mga circuit breaker, mga limitasyon sa rate -- `/dashboard/logs` — kahilingan/proxy/audit/console log -- `/dashboard/settings` — mga tab ng mga setting ng system (pangkalahatan, pagruruta, mga default ng combo, atbp.) -- `/dashboard/api-manager` — Lifecycle ng key ng API at mga pahintulot ng modelo## High-Level System Context +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Mga pangunahing direktoryo: +Main directories: -- `src/app/api/v1/*` at `src/app/api/v1beta/*` para sa mga compatibility API -- `src/app/api/*` para sa mga management/configuration API -- Susunod na muling pagsusulat sa `next.config.mjs` na mapa `/v1/*` sa `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Mahahalagang ruta ng compatibility: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` — kasama ang mga custom na modelo na may `custom: true` -- `src/app/api/v1/embeddings/route.ts` — pagbuo ng pag-embed (6 na provider) -- `src/app/api/v1/images/generations/route.ts` — pagbuo ng larawan (4+ provider kasama ang Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — nakatuong per-provider chat -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — nakalaang mga pag-embed ng bawat provider -- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — nakalaang mga larawan ng bawat provider +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` - `src/app/api/v1beta/models/[...path]/route.ts` -Mga domain ng pamamahala: +Management domains: -- Auth/setting: `src/app/api/auth/*`, `src/app/api/settings/*` -- Mga provider/koneksyon: `src/app/api/providers*` -- Mga node ng provider: `src/app/api/provider-nodes*` -- Mga custom na modelo: `src/app/api/provider-models` (GET/POST/DELETE) -- Catalog ng modelo: `src/app/api/models/route.ts` (GET) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) - Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- Mga Key/aliases/combos/presyo: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- Paggamit: `src/app/api/usage/*` -- Pag-sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` -- Mga katulong sa CLI tooling: `src/app/api/cli-tools/*` +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` - IP filter: `src/app/api/settings/ip-filter` (GET/PUT) -- Pag-iisip na badyet: `src/app/api/settings/thinking-budget` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) - System prompt: `src/app/api/settings/system-prompt` (GET/PUT) -- Mga Session: `src/app/api/sessions` (GET) -- Mga limitasyon sa rate: `src/app/api/rate-limits` (GET) -- Katatagan: `src/app/api/resilience` (GET/PATCH) — mga profile ng provider, circuit breaker, estado ng limitasyon sa rate +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state - Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns -- Mga istatistika ng cache: `src/app/api/cache/stats` (GET/DELETE) -- Availability ng modelo: `src/app/api/models/availability` (GET/POST) +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) - Telemetry: `src/app/api/telemetry/summary` (GET) -- Badyet: `src/app/api/usage/budget` (GET/POST) -- Fallback chain: `src/app/api/fallback/chains` (GET/POST/DELETE) -- Pag-audit sa pagsunod: `src/app/api/compliance/audit-log` (GET) -- Mga Eval: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- Mga Patakaran: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) + +## 2) SSE + Translation Core Main flow modules: - Entry: `src/sse/handlers/chat.ts` -- Pangunahing orkestrasyon: `open-sse/handlers/chatCore.ts` -- Mga adaptor ng pagpapatupad ng provider: `open-sse/executors/*` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` - Format detection/provider config: `open-sse/services/provider.ts` - Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Logic ng fallback ng account: `open-sse/services/accountFallback.ts` -- Rehistro ng pagsasalin: `open-sse/translator/index.ts` -- Mga pagbabago sa stream: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- Pagkuha/normalisasyon ng paggamit: `open-sse/utils/usageTracking.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` - Think tag parser: `open-sse/utils/thinkTagParser.ts` -- Handler ng pag-embed: `open-sse/handlers/embeddings.ts` -- Pag-embed ng registry ng provider: `open-sse/config/embeddingRegistry.ts` -- Handler ng pagbuo ng larawan: `open-sse/handlers/imageGeneration.ts` -- Rehistro ng provider ng larawan: `open-sse/config/imageRegistry.ts` -- Paglilinis ng tugon: `open-sse/handlers/responseSanitizer.ts` -- Pag-normalize ng tungkulin: `open-sse/services/roleNormalizer.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -Mga Serbisyo (lohika ng negosyo): +Services (business logic): -- Pagpili/pagmamarka ng account: `open-sse/services/accountSelector.ts` -- Pamamahala ng lifecycle ng konteksto: `open-sse/services/contextManager.ts` -- Pagpapatupad ng IP filter: `open-sse/services/ipFilter.ts` -- Pagsubaybay sa session: `open-sse/services/sessionManager.ts` -- Humiling ng deduplication: `open-sse/services/signatureCache.ts` +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` - System prompt injection: `open-sse/services/systemPrompt.ts` -- Pag-iisip ng pamamahala ng badyet: `open-sse/services/thinkingBudget.ts` -- Pagruruta ng modelo ng wildcard: `open-sse/services/wildcardRouter.ts` -- Pamamahala sa limitasyon ng rate: `open-sse/services/rateLimitManager.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` - Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -Mga module ng layer ng domain: +Domain layer modules: -- Availability ng modelo: `src/lib/domain/modelAvailability.ts` -- Mga panuntunan/badyet ng gastos: `src/lib/domain/costRules.ts` -- Patakaran sa Fallback: `src/lib/domain/fallbackPolicy.ts` +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` - Combo resolver: `src/lib/domain/comboResolver.ts` -- Patakaran sa pag-lockout: `src/lib/domain/lockoutPolicy.ts` -- Policy engine: `src/domain/policyEngine.ts` — sentralisadong lockout → badyet → fallback evaluation -- Catalog ng mga error code: `src/lib/domain/errorCodes.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` - Request ID: `src/lib/domain/requestId.ts` -- I-fetch ang timeout: `src/lib/domain/fetchTimeout.ts` -- Humiling ng telemetry: `src/lib/domain/requestTelemetry.ts` -- Pagsunod/pag-audit: `src/lib/domain/compliance/index.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` - Eval runner: `src/lib/domain/evalRunner.ts` -- Pananatili ng estado ng domain: `src/lib/db/domainState.ts` — SQLite CRUD para sa mga fallback na chain, badyet, kasaysayan ng gastos, estado ng lockout, mga circuit breaker +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -Mga module ng OAuth provider (12 indibidwal na file sa ilalim ng `src/lib/oauth/providers/`): +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): - Registry index: `src/lib/oauth/providers/index.ts` -- Mga indibidwal na provider: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts.`, `kilo`c.ts. -- Manipis na wrapper: `src/lib/oauth/providers.ts` — muling pag-export mula sa mga indibidwal na module## 3) Persistence Layer +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -Pangunahing estado DB (SQLite): +## 3) Persistence Layer + +Primary state DB (SQLite): - Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) -- Muling i-export ang facade: `src/lib/localDb.ts` (manipis na layer ng compatibility para sa mga tumatawag) -- file: `${DATA_DIR}/storage.sqlite` (o `$XDG_CONFIG_HOME/omniroute/storage.sqlite` kapag nakatakda, kung hindi `~/.omniroute/storage.sqlite`) -- mga entity (table + KV namespaces): providerConnections, providerNodes, modelAliases, combo, apiKeys, setting, pagpepresyo,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -Pagtitiyaga ng paggamit: +Usage persistence: -- facade: `src/lib/usageDb.ts` (mga decomposed modules sa `src/lib/usage/*`) -- Mga talahanayan ng SQLite sa `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` -- nananatili ang mga opsyonal na artifact ng file para sa compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- Ang mga legacy na JSON file ay inililipat sa SQLite sa pamamagitan ng mga startup na paglilipat kapag naroroon +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present Domain State DB (SQLite): -- `src/lib/db/domainState.ts` — Mga pagpapatakbo ng CRUD para sa estado ng domain -- Mga talahanayan (ginawa sa `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- Write-through na cache pattern: in-memoryang Maps ay may awtoridad sa runtime; ang mga mutasyon ay nakasulat nang sabay-sabay sa SQLite; ang estado ay naibalik mula sa DB sa malamig na simula## 4) Auth + Security Surfaces +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces - Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- Pagbuo/pag-verify ng API key: `src/shared/utils/apiKey.ts` -- Nagpatuloy ang mga lihim ng provider sa mga entry ng `providerConnections` -- Outbound proxy na suporta sa pamamagitan ng `open-sse/utils/proxyFetch.ts` (env vars) at `open-sse/utils/networkProxy.ts` (configurable per-provider o global)## 5) Cloud Sync +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync - Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Pana-panahong gawain: `src/shared/services/cloudSyncScheduler.ts` -- Pana-panahong gawain: `src/shared/services/modelSyncScheduler.ts` -- Kontrolin ang ruta: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Ang mga desisyon sa Fallback ay hinihimok ng `open-sse/services/accountFallback.ts` gamit ang mga status code at error-message heuristics. Ang combo routing ay nagdaragdag ng isang karagdagang bantay: provider-scoped 400s gaya ng upstream content-block at role-validation failures ay itinuturing bilang model-local na mga pagkabigo kaya maaaring tumakbo pa rin ang mga combo target sa ibang pagkakataon.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -Ang pag-refresh sa panahon ng live na trapiko ay isinasagawa sa loob ng `open-sse/handlers/chatCore.ts` sa pamamagitan ng executor `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -Ang pana-panahong pag-sync ay na-trigger ng `CloudSyncScheduler` kapag pinagana ang cloud.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -Mga file ng pisikal na storage: +Physical storage files: -- pangunahing runtime DB: `${DATA_DIR}/storage.sqlite` -- mga linya ng log ng kahilingan: `${DATA_DIR}/log.txt` (compat/debug artifact) -- structured call payload archive: `${DATA_DIR}/call_logs/` -- opsyonal na tagasalin/paghiling ng mga sesyon ng pag-debug: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: mga compatibility API -- `src/app/api/v1/providers/[provider]/*`: nakalaang mga ruta ng bawat provider (chat, mga pag-embed, mga larawan) +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) - `src/app/api/providers*`: provider CRUD, validation, testing -- `src/app/api/provider-nodes*`: custom na katugmang pamamahala ng node -- `src/app/api/provider-models`: pamamahala ng custom na modelo (CRUD) -- `src/app/api/models/route.ts`: model catalog API (mga alias + custom na modelo) -- `src/app/api/oauth/*`: Ang mga daloy ng OAuth/device-code -- `src/app/api/keys*`: lokal na API key lifecycle -- `src/app/api/models/alias`: pamamahala ng alias -- `src/app/api/combos*`: pamamahala ng fallback combo -- `src/app/api/pricing`: na-override ang pagpepresyo para sa pagkalkula ng gastos -- `src/app/api/settings/proxy`: configuration ng proxy (GET/PUT/DELETE) +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) - `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) -- `src/app/api/usage/*`: mga API ng paggamit at mga log -- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync at cloud-facing helper -- `src/app/api/cli-tools/*`: mga lokal na CLI config writers/checkers +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers - `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) - `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) - `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) -- `src/app/api/sessions`: aktibong listahan ng session (GET) -- `src/app/api/rate-limits`: per-account rate limit status (GET)### Routing and Execution Core +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: humiling ng parse, combo handling, loop ng pagpili ng account -- `open-sse/handlers/chatCore.ts`: pagsasalin, executor dispatch, retry/refresh handling, stream setup -- `open-sse/executors/*`: network na partikular sa provider at gawi sa format### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: registry ng translator at orkestrasyon -- Humiling ng mga tagasalin: `open-sse/translator/request/*` -- Mga tagasalin ng tugon: `open-sse/translator/response/*` -- Mga constant ng format: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: persistent config/state at domain persistence sa SQLite -- `src/lib/localDb.ts`: muling pag-export ng compatibility para sa mga DB module -- `src/lib/usageDb.ts`: kasaysayan ng paggamit/mga log ng tawag na facade sa itaas ng mga talahanayan ng SQLite## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Ang bawat provider ay may dalubhasang tagapagpatupad na nagpapalawak ng `BaseExecutor` (sa `open-sse/executors/base.ts`), na nagbibigay ng pagbuo ng URL, pagbuo ng header, muling subukang may exponential backoff, credential refresh hook, at ang `execute()` na paraan ng orkestrasyon. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Tagapagpatupad | (Mga) Provider | Espesyal na Paghawak | -| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------------------------------------------------------- | -| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic na URL/header config bawat provider | -| `AntigravityExecutor` | Google Antigravity | Mga custom na project/session ID, Retry-After parsing | -| `CodexExecutor` | OpenAI Codex | Nag-inject ng mga tagubilin sa system, pinipilit ang pagsisikap sa pangangatwiran | -| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, kahilingan sa pagpirma sa pamamagitan ng checksum | -| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking header | -| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | -| `GeminiCLIEexecutor` | Gemini CLI | Ikot ng pag-refresh ng token ng Google OAuth | +### Persistence -Ang lahat ng iba pang provider (kabilang ang mga custom na katugmang node) ay gumagamit ng `DefaultExecutor`.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Provider | Format | Awto | Stream | Hindi Stream | Pag-refresh ng Token | Paggamit ng API | -| ---------------- | ---------------- | --------------------- | ---------------- | ------------ | -------------------- | ----------------------------- | ------------------------------ | -| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin lang | -| Gemini | Gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | -| Gemini CLI | Gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | -| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Buong quota API | -| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | -| Codex | openai-responses | OAuth | ✅ pinilit | ❌ | ✅ | ✅ Mga limitasyon sa rate | -| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Mga snapshot ng quota | -| Cursor | cursor | Custom na checksum | ✅ | ✅ | ❌ | ❌ | -| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Mga limitasyon sa paggamit | -| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Bawat kahilingan | -| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Bawat kahilingan | -| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | -| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | -| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | -| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | -| Pagkagulo | openai | API Key | ✅ | ✅ | ❌ | ❌ | -| Magkasama AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | -| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | -| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | -| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Kasama sa mga natukoy na format ng pinagmulan ang: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. + +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | + +All other providers (including custom compatible nodes) use the `DefaultExecutor`. + +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: - `openai` - `openai-responses` - `claude` - `gemini` -Kasama sa mga target na format ang: +Target formats include: -- OpenAI chat/Mga Tugon +- OpenAI chat/Responses - Claude - Gemini/Gemini-CLI/Antigravity envelope - Kiro - Cursor -Ginagamit ng mga pagsasalin ang**OpenAI bilang hub format**— lahat ng conversion ay dumadaan sa OpenAI bilang intermediate:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Pinipili ang mga pagsasalin sa dynamic na paraan batay sa hugis ng source payload at format ng target ng provider. +Additional processing layers in the translation pipeline: -Mga karagdagang layer ng pagpoproseso sa pipeline ng pagsasalin: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Response sanitization**— Tinatanggal ang mga hindi karaniwang field mula sa OpenAI-format na mga tugon (parehong streaming at non-streaming) para matiyak ang mahigpit na pagsunod sa SDK --**Pag-normalize ng tungkulin**— Kino-convert ang `developer` → `system` para sa mga target na hindi OpenAI; pinagsasama ang `system` → `user` para sa mga modelong tumatanggi sa papel ng system (GLM, ERNIE) --**Think tag extraction**— Pina-parse ang `...` block mula sa content papunta sa `reasoning_content` field --**Structured output**— Kino-convert ang OpenAI `response_format.json_schema` sa `responseMimeType` + `responseSchema` ni Gemini## Supported API Endpoints +## Supported API Endpoints -| Endpoint | Format | Handler | -| ---------------------------------------------------- | ------------------- | ------------------------------------------------------------------- | -| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | -| `POST /v1/messages` | Claude Messages | Parehong handler (auto-detected) | -| `POST /v1/mga tugon` | Mga Tugon sa OpenAI | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | -| `GET /v1/embeddings` | Listahan ng modelo | ruta ng API | -| `POST /v1/images/generations` | Mga Larawan ng OpenAI | `open-sse/handlers/imageGeneration.ts` | -| `GET /v1/images/generations` | Listahan ng modelo | ruta ng API | -| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Nakatuon sa bawat provider na may pagpapatunay ng modelo | -| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Nakatuon sa bawat provider na may pagpapatunay ng modelo | -| `POST /v1/providers/{provider}/images/generations` | Mga Larawan ng OpenAI | Nakatuon sa bawat provider na may pagpapatunay ng modelo | -| `POST /v1/messages/count_tokens` | Bilang ng Token ng Claude | ruta ng API | -| `GET /v1/models` | Listahan ng OpenAI Models | ruta ng API (chat + pag-embed + larawan + mga custom na modelo) | -| `GET /api/models/catalog` | Catalog | Lahat ng mga modelo ay nakapangkat ayon sa provider + uri | -| `POST /v1beta/models/*:streamGenerateContent` | Taong Gemini | ruta ng API | -| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Configuration ng proxy ng network | -| `POST /api/settings/proxy/test` | Pagkakakonekta ng Proxy | Endpoint ng pagsubok sa kalusugan/pagkonekta ng proxy | -| `GET/POST/DELETE /api/provider-models` | Mga Modelo ng Provider | Mga custom at pinamamahalaang available na modelo na sinusuportahan ng metadata ng modelo ng provider |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -Hinaharang ng bypass handler (`open-sse/utils/bypassHandler.ts`) ang mga kilalang "throwaway" na kahilingan mula kay Claude CLI — mga warmup ping, pagkuha ng pamagat, at mga bilang ng token — at nagbabalik ng**pekeng tugon**nang hindi gumagamit ng upstream na mga token ng provider. Nati-trigger lang ito kapag ang `User-Agent` ay naglalaman ng `claude-cli`.## Request Logger Pipeline +## Bypass Handler -Ang request logger (`open-sse/utils/requestLogger.ts`) ay nagbibigay ng 7-stage na debug logging pipeline, na hindi pinagana bilang default, na pinagana sa pamamagitan ng `ENABLE_REQUEST_LOGS=true`:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Ang mga file ay isinusulat sa `/logs//` para sa bawat sesyon ng kahilingan.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- cooldown ng provider account sa mga lumilipas/rate/auth error -- fallback ng account bago mabigo ang kahilingan -- fallback ng combo model kapag naubos na ang kasalukuyang modelo/provider path## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- paunang suriin at i-refresh na may muling pagsubok para sa mga nare-refresh na provider -- 401/403 muling subukan pagkatapos i-refresh ang pagtatangka sa pangunahing landas## 3) Stream Safety +## 2) Token Expiry + +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path + +## 3) Stream Safety - disconnect-aware stream controller -- translation stream na may end-of-stream flush at `[DONE]` handling -- fallback sa pagtatantya ng paggamit kapag nawawala ang metadata ng paggamit ng provider## 4) Cloud Sync Degradation +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -- Lumilitaw ang mga error sa pag-sync ngunit nagpapatuloy ang lokal na runtime -- Ang scheduler ay may retry-capable logic, ngunit ang pana-panahong execution ay kasalukuyang tumatawag sa single-attempt sync bilang default## 5) Data Integrity +## 4) Cloud Sync Degradation -- Mga paglilipat ng schema ng SQLite at mga auto-upgrade na hook sa pagsisimula -- legacy na JSON → path ng compatibility ng paglilipat ng SQLite## Observability and Operational Signals +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -Runtime visibility source: +## 5) Data Integrity -- mga console log mula sa `src/sse/utils/logger.ts` -- mga pinagsama-samang paggamit sa bawat kahilingan sa SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- apat na yugto ng detalyadong payload na kumukuha sa SQLite (`request_detail_logs`) kapag `settings.detailed_logs_enabled=true` -- log in sa status ng textual na kahilingan `log.txt` (opsyonal/compat) -- opsyonal na malalim na kahilingan/mga log ng pagsasalin sa ilalim ng `mga log/` kapag `ENABLE_REQUEST_LOGS=true` -- Mga endpoint sa paggamit ng dashboard (`/api/usage/*`) para sa paggamit ng UI +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -Ang detalyadong paghiling ng payload capture ay nag-iimbak ng hanggang apat na JSON payload stages sa bawat rutang tawag: +## Observability and Operational Signals -- hilaw na kahilingan na natanggap mula sa kliyente -- isinalin na kahilingan ay aktwal na ipinadala sa upstream -- ang tugon ng provider ay muling itinayo bilang JSON; ang mga naka-stream na tugon ay pinagsama sa panghuling buod at stream metadata -- panghuling tugon ng kliyente na ibinalik ng OmniRoute; naka-imbak ang mga naka-stream na tugon sa parehong compact summary form## Security-Sensitive Boundaries +Runtime visibility sources: -- Sikreto ng JWT (`JWT_SECRET`) ay sinisiguro ang pag-verify/pagpirma ng cookie ng session ng dashboard -- Ang paunang password bootstrap (`INITIAL_PASSWORD`) ay dapat na tahasang i-configure para sa first-run provisioning -- Ang API key HMAC secret (`API_KEY_SECRET`) ay sinisiguro ang nabuong lokal na format ng API key -- Ang mga lihim ng provider (mga API key/token) ay nananatili sa lokal na DB at dapat na protektahan sa antas ng filesystem -- Umaasa ang mga endpoint ng cloud sync sa API key auth + semantics ng machine id## Environment and Runtime Matrix +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -Mga variable ng kapaligiran na aktibong ginagamit ng code: +Detailed request payload capture stores up to four JSON payload stages per routed call: + +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: - App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` -- Imbakan: `DATA_DIR` -- Katugmang pag-uugali ng node: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Opsyonal na pag-override sa base ng imbakan (Linux/macOS kapag hindi nakatakda ang `DATA_DIR`): `XDG_CONFIG_HOME` -- Pag-hash ng seguridad: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Pag-log: `ENABLE_REQUEST_LOGS` -- Pag-sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` at lowercase na variant -- Mga flag ng tampok ng SOCKS5: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- Mga katulong sa platform/runtime (hindi config na partikular sa app): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` -1. Ang `usageDb` at `localDb` ay nagbabahagi ng parehong base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) na may legacy na paglipat ng file. -2. Nagde-delegate ang `/api/v1/route.ts` sa parehong pinag-isang tagabuo ng catalog na ginagamit ng `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) upang maiwasan ang semantic drift. -3. Ang Request logger ay nagsusulat ng buong header/body kapag pinagana; ituring ang direktoryo ng log bilang sensitibo. -4. Nakadepende ang pag-uugali ng cloud sa tamang `NEXT_PUBLIC_BASE_URL` at maabot ng cloud endpoint. -5. Ang direktoryo ng `open-sse/` ay na-publish bilang `@omniroute/open-sse`**npm workspace package**. Ini-import ito ng source code sa pamamagitan ng `@omniroute/open-sse/...` (nalutas ng Next.js `transpilePackages`). Ginagamit pa rin ng mga path ng file sa dokumentong ito ang pangalan ng direktoryo na `open-sse/` para sa pagkakapare-pareho. -6. Ang mga chart sa dashboard ay gumagamit ng**Recharts**(SVG-based) para sa naa-access, interactive na mga visualization ng analytics (mga bar chart ng paggamit ng modelo, mga talahanayan ng breakdown ng provider na may mga rate ng tagumpay). -7. Gumagamit ang E2E tests ng**Playwright**(`tests/e2e/`), na tumatakbo sa pamamagitan ng `npm run test:e2e`. Gumagamit ang mga unit test ng**Node.js test runner**(`mga pagsubok/unit/`), tumatakbo sa pamamagitan ng `npm run test:unit`. Ang source code sa ilalim ng `src/` ay**TypeScript**(`.ts`/`.tsx`); ang `open-sse/` workspace ay nananatiling JavaScript (`.js`). -8. Ang pahina ng mga setting ay isinaayos sa 5 tab: Seguridad, Pagruruta (6 na pandaigdigang diskarte: fill-first, round-robin, p2c, random, hindi gaanong ginagamit, cost-optimized), Resilience (editable rate limits, circuit breaker, mga patakaran), AI (thinking budget, system prompt, prompt cache), Advanced (proxy).## Operational Verification Checklist +## Known Architectural Notes -- Bumuo mula sa pinagmulan: `npm run build` -- Bumuo ng imahe ng Docker: `docker build -t omniroute .` -- Simulan ang serbisyo at i-verify: +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: - `GET /api/settings` - `GET /api/v1/models` -- Ang CLI target base URL ay dapat na `http://:20128/v1` kapag `PORT=20128` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/phi/docs/FEATURES.md b/docs/i18n/phi/docs/FEATURES.md index 98b6cc342c..fc8c2a8096 100644 --- a/docs/i18n/phi/docs/FEATURES.md +++ b/docs/i18n/phi/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Visual na gabay sa bawat seksyon ng OmniRoute dashboard.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Pamahalaan ang mga koneksyon sa AI provider: OAuth provider (Claude Code, Codex, Gemini CLI), API key provider (Groq, DeepSeek, OpenRouter), at libreng provider (Qoder, Qwen, Kiro). Kasama sa mga Kiro account ang pagsubaybay sa balanse ng kredito — mga natitirang credit, kabuuang allowance, at petsa ng pag-renew na makikita sa Dashboard → Paggamit.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Gumawa ng mga combo sa pagruruta ng modelo na may 6 na diskarte: priority, weighted, round-robin, random, hindi gaanong ginagamit, at cost-optimized. Ang bawat combo ay nagkakadena ng maraming modelo na may awtomatikong fallback at may kasamang mabilis na mga template at mga pagsusuri sa kahandaan.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Komprehensibong analytics ng paggamit na may pagkonsumo ng token, mga pagtatantya sa gastos, mga heatmap ng aktibidad, lingguhang chart ng pamamahagi, at mga breakdown sa bawat provider.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Real-time na pagsubaybay: uptime, memorya, bersyon, latency percentiles (p50/p95/p99), mga istatistika ng cache, at mga estado ng circuit breaker ng provider.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Apat na mode para sa pag-debug ng mga pagsasalin ng API:**Playground**(format converter),**Chat Tester**(live na kahilingan),**Test Bench**(batch tests), at**Live Monitor**(real-time stream).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Subukan ang anumang modelo nang direkta mula sa dashboard. Pumili ng provider, modelo, at endpoint, magsulat ng mga prompt gamit ang Monaco Editor, mag-stream ng mga tugon sa real-time, i-abort ang mid-stream, at tingnan ang mga sukatan ng timing.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Nako-customize na mga tema ng kulay para sa buong dashboard. Pumili mula sa 7 preset na kulay (Coral, Blue, Red, Green, Violet, Orange, Cyan) o gumawa ng custom na tema sa pamamagitan ng pagpili ng anumang hex na kulay. Sinusuportahan ang liwanag, madilim, at system mode.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Panel ng kumpletong mga setting na may mga tab: +Comprehensive settings panel with tabs: --**General**— System storage, backup management (export/import database) -**Hitsura**— Tagapili ng tema (madilim/liwanag/system), mga preset ng tema ng kulay at mga custom na kulay, visibility ng log ng kalusugan, mga kontrol sa visibility ng item sa sidebar -**Seguridad**— Proteksyon ng endpoint ng API, custom na pagharang ng provider, pag-filter ng IP, impormasyon ng session -**Pagruruta**— Mga alyas ng modelo, pagkasira ng gawain sa background -**Resilience**— Pagpapatuloy ng limitasyon sa rate, pag-tune ng circuit breaker, awtomatikong i-disable ang mga naka-ban na account, pagsubaybay sa expiration ng provider -**Advanced**— Mga override sa configuration, configuration audit trail, fallback degradation mode![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Isang-click na configuration para sa AI coding tool: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, at Factory Droid. Nagtatampok ng awtomatikong paglalapat/pag-reset ng config, mga profile ng koneksyon, at pagmamapa ng modelo.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Dashboard para sa pagtuklas at pamamahala ng mga ahente ng CLI. Nagpapakita ng grid ng 14 na built-in na ahente (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) na may: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Katayuan ng pag-install**— Naka-install / Hindi Natagpuan na may pagtukoy ng bersyon -**Protocol badge**— stdio, HTTP, atbp. -**Mga custom na ahente**— Magrehistro ng anumang CLI tool sa pamamagitan ng form (pangalan, binary, version command, spawn args) -**Pagtutugma ng CLI Fingerprint**— Toggle ng bawat provider upang tumugma sa mga native na lagda ng kahilingan sa CLI, na binabawasan ang panganib sa pagbabawal habang pinapanatili ang proxy IP--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Bumuo ng mga larawan, video, at musika mula sa dashboard. Sinusuportahan ang OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, at MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Real-time na pag-log ng kahilingan gamit ang pag-filter ayon sa provider, modelo, account, at API key. Nagpapakita ng mga status code, paggamit ng token, latency, at mga detalye ng pagtugon.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Ang iyong pinag-isang API endpoint na may pagkasira ng kakayahan: Mga Pagkumpleto ng Chat, Mga Tugon na API, Mga Pag-embed, Pagbuo ng Larawan, Muling Ranggo, Transkripsyon ng Audio, Text-to-Speech, Mga Moderation, at mga nakarehistrong API key. Cloudflare Quick Tunnel integration at suporta sa cloud proxy para sa malayuang pag-access.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -Gumawa, saklaw, at bawiin ang mga API key. Ang bawat key ay maaaring paghigpitan sa mga partikular na modelo/provider na may ganap na access o read-only na mga pahintulot. Pamamahala ng visual key na may pagsubaybay sa paggamit.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Pagsubaybay sa administratibong pagkilos na may pag-filter ayon sa uri ng pagkilos, aktor, target, IP address, at timestamp. Buong kasaysayan ng kaganapan sa seguridad.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Native Electron desktop app para sa Windows, macOS, at Linux. Patakbuhin ang OmniRoute bilang isang standalone na application na may system tray integration, offline na suporta, auto-update, at one-click na pag-install. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Mga pangunahing tampok: +Key features: -- Pagboto sa kahandaan ng server (walang blangkong screen sa malamig na simula) -- System tray na may port management -- Patakaran sa Seguridad ng Nilalaman +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy - Single-instance lock -- Auto-update sa pag-restart +- Auto-update on restart - Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) -- Hardened Electron build packaging — ang mga naka-symlink na `node_modules` sa standalone na bundle ay nakita at tinanggihan bago ang packaging, na pumipigil sa runtime dependency sa build machine (v2.5.5+) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Tingnan ang [`electron/README.md`](../electron/README.md) para sa buong dokumentasyon. +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/phi/docs/TROUBLESHOOTING.md b/docs/i18n/phi/docs/TROUBLESHOOTING.md index 87ed64090e..d61631edc5 100644 --- a/docs/i18n/phi/docs/TROUBLESHOOTING.md +++ b/docs/i18n/phi/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -Mga karaniwang problema at solusyon para sa OmniRoute.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Problema | Solusyon | -| ------------------------------------------------ | ------------------------------------------------------------------------------ | --- | -| Unang login ay hindi gumagana | Itakda ang `INITIAL_PASSWORD` sa `.env` (walang hardcoded default) | -| Nagbubukas ang dashboard sa maling port | Itakda ang `PORT=20128` at `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Walang mga log ng kahilingan sa ilalim ng `log/` | Itakda ang `ENABLE_REQUEST_LOGS=true` | -| EACCES: tinanggihan ang pahintulot | Itakda ang `DATA_DIR=/path/to/writable/dir` para i-override ang `~/.omniroute` | -| Hindi nagse-save ang diskarte sa pagruruta | Update sa v1.4.11+ (Zod schema fix para sa pagtitiyaga ng mga setting) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Sanhi:**Naubos na ang quota ng provider. +**Cause:** Provider quota exhausted. -**Ayusin:** +**Fix:** -1. Suriin ang dashboard quota tracker -2. Gumamit ng combo na may fallback tier -3. Lumipat sa mas mura/libreng tier### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Dahil:**Naubos na ang quota ng subscription. +### Rate Limiting -**Ayusin:** +**Cause:** Subscription quota exhausted. -- Magdagdag ng fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Gamitin ang GLM/MiniMax bilang murang backup### OAuth Token Expired +**Fix:** -Ang OmniRoute ay awtomatikong nagre-refresh ng mga token. Kung magpapatuloy ang mga isyu: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Dashboard → Provider → Kumonekta muli -2. Tanggalin at muling idagdag ang koneksyon ng provider--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. I-verify ang mga `BASE_URL` na puntos sa iyong running instance (hal., `http://localhost:20128`) -2. I-verify ang mga `CLOUD_URL` na puntos sa iyong cloud endpoint (hal., `https://omniroute.dev`) -3. Panatilihing nakahanay ang mga value ng `NEXT_PUBLIC_*` sa mga value sa gilid ng server### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Symptom:**`Hindi inaasahang token 'd'...` sa cloud endpoint para sa mga non-streaming na tawag. +### Cloud `stream=false` Returns 500 -**Sanhi:**Ibinabalik ng Upstream ang SSE payload habang inaasahan ng kliyente ang JSON. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Workaround:**Gamitin ang `stream=true` para sa mga direktang tawag sa cloud. Kasama sa lokal na runtime ang SSE→JSON fallback.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Gumawa ng bagong key mula sa lokal na dashboard (`/api/keys`) -2. Patakbuhin ang cloud sync: Paganahin ang Cloud → Sync Now -3. Ang mga luma/hindi naka-sync na key ay maaari pa ring ibalik ang `401` sa cloud--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Suriin ang mga field ng runtime: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. Para sa portable mode: gumamit ng target ng imahe na `runner-cli` (mga naka-bundle na CLI) -3. Para sa host mount mode: itakda ang `CLI_EXTRA_PATHS` at i-mount ang host bin directory bilang read-only -4. Kung `naka-install=true` at `runnable=false`: nakita ang binary ngunit nabigo ang healthcheck### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Suriin ang mga istatistika ng paggamit sa Dashboard → Paggamit -2. Ilipat ang pangunahing modelo sa GLM/MiniMax -3. Gumamit ng libreng tier (Gemini CLI, Qoder) para sa mga hindi kritikal na gawain -4. Magtakda ng mga badyet sa gastos sa bawat API key: Dashboard → API Keys → Badyet--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Itakda ang `ENABLE_REQUEST_LOGS=true` sa iyong `.env` file. Lumilitaw ang mga log sa ilalim ng direktoryo ng `log/`.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Pangunahing estado: `${DATA_DIR}/storage.sqlite` (mga provider, combo, alias, key, setting) -- Paggamit: Mga talahanayan ng SQLite sa `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + opsyonal na `${DATA_DIR}/log.txt` at `${DATA_DIR}/call_logs/` -- Mga log ng kahilingan: `/logs/...` (kapag `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Kapag ang circuit breaker ng provider ay BUKAS, ang mga kahilingan ay hinaharangan hanggang sa mag-expire ang cooldown. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Ayusin:** +**Fix:** -1. Pumunta sa**Dashboard → Settings → Resilience** -2. Suriin ang circuit breaker card para sa apektadong provider -3. I-click ang**I-reset Lahat**upang i-clear ang lahat ng mga breaker, o hintaying mag-expire ang cooldown -4. I-verify na available talaga ang provider bago i-reset### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Kung ang isang provider ay paulit-ulit na pumasok sa OPEN state: +### Provider keeps tripping the circuit breaker -1. Suriin ang**Dashboard → Health → Provider Health**para sa pattern ng pagkabigo -2. Pumunta sa**Settings → Resilience → Provider Profiles**at taasan ang failure threshold -3. Suriin kung binago ng provider ang mga limitasyon ng API o nangangailangan ng muling pagpapatotoo -4. Suriin ang latency telemetry — ang mataas na latency ay maaaring magdulot ng mga pagkabigo batay sa timeout--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Tiyaking ginagamit mo ang tamang prefix: `deepgram/nova-3` o `assemblyai/best` -- I-verify na konektado ang provider sa**Dashboard → Mga Provider**### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Suriin ang mga sinusuportahang format ng audio: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- I-verify na ang laki ng file ay nasa loob ng mga limitasyon ng provider (karaniwang <25MB) -- Suriin ang validity ng provider ng API key sa provider card--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Gamitin ang**Dashboard → Translator**upang i-debug ang mga isyu sa pagsasalin ng format: +Use **Dashboard → Translator** to debug format translation issues: -| Mode | Kailan Gagamitin | -| ---------------- | ----------------------------------------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Laruan** | Paghambingin ang mga format ng input/output nang magkatabi — i-paste ang isang nabigong kahilingan upang makita kung paano ito isinasalin | -| **Chat Tester** | Magpadala ng mga live na mensahe at siyasatin ang buong kahilingan/tugon payload kasama ang mga header | -| **Test Bench** | Magpatakbo ng mga batch test sa mga kumbinasyon ng format upang malaman kung aling mga pagsasalin ang sira | -| **Live Monitor** | Panoorin ang daloy ng kahilingan sa real-time upang mahuli ang mga pasulput-sulpot na isyu sa pagsasalin | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Hindi lumalabas ang mga tag ng pag-iisip**— Tingnan kung sinusuportahan ng target na provider ang pag-iisip at ang setting ng badyet sa pag-iisip -**Pagbaba ng mga tawag sa tool**— Maaaring alisin ng ilang pagsasalin ng format ang mga hindi sinusuportahang field; i-verify sa Playground mode -**System prompt nawawala**— Claude at Gemini handle system prompts magkaiba; suriin ang output ng pagsasalin -**Nagbabalik ang SDK ng raw string sa halip na object**— Naayos sa v1.1.0: tinatanggal na ngayon ng response sanitizer ang mga hindi karaniwang field (`x_groq`, `usage_breakdown`, atbp.) na nagdudulot ng mga pagkabigo sa pagpapatunay ng OpenAI SDK Pydantic -**Tinatanggihan ng GLM/ERNIE ang tungkulin ng `system`**— Naayos sa v1.1.0: awtomatikong pinagsasama ng role normalizer ang mga mensahe ng system sa mga mensahe ng user para sa mga hindi tugmang modelo -**Hindi nakilala ang tungkulin ng `developer`**— Naayos sa v1.1.0: awtomatikong na-convert sa `system` para sa mga provider na hindi OpenAI -**`json_schema` hindi gumagana sa Gemini**— Naayos sa v1.1.0: `response_format` ay na-convert na ngayon sa `responseMimeType` + `responseSchema` ng Gemini--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- Nalalapat lang ang limitasyon ng awtomatikong rate sa mga provider ng API key (hindi OAuth/subscription) -- I-verify**Mga Setting → Resilience → Provider Profile**ay pinagana ang auto-rate-limit -- Suriin kung ang provider ay nagbabalik ng `429` status code o `Retry-After` header### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -Sinusuportahan ng mga profile ng provider ang mga setting na ito: +### Tuning exponential backoff --**Base delay**— Paunang oras ng paghihintay pagkatapos ng unang pagkabigo (default: 1s) -**Max na pagkaantala**— Maximum na limitasyon sa oras ng paghihintay (default: 30s) -**Multiplier**— Magkano ang itataas na pagkaantala sa bawat magkakasunod na pagkabigo (default: 2x)### Anti-thundering herd +Provider profiles support these settings: -Kapag maraming sabay-sabay na kahilingan ang tumama sa isang provider na limitado sa rate, gumagamit ang OmniRoute ng mutex + auto rate-limiting para i-serialize ang mga kahilingan at maiwasan ang mga pagkabigo ng cascading. Ito ay awtomatiko para sa mga API key provider.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Inilalagay ng ilang user ng OmniRoute ang gateway sa harap ng RAG o mga stack ng ahente. Sa mga setup na iyon, karaniwan nang makakita ng kakaibang pattern: Mukhang malusog ang OmniRoute (mga provider up, ok ang mga profile sa pagruruta, walang mga alerto sa limitasyon sa rate) ngunit mali pa rin ang huling sagot. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -Sa pagsasagawa, ang mga insidenteng ito ay karaniwang nagmumula sa downstream na RAG pipeline, hindi mula sa gateway mismo. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Kung gusto mo ng nakabahaging bokabularyo upang ilarawan ang mga pagkabigo na iyon, maaari mong gamitin ang WFGY ProblemMap, isang panlabas na mapagkukunan ng teksto ng lisensya ng MIT na tumutukoy sa labing-anim na umuulit na pattern ng pagkabigo ng RAG / LLM. Sa isang mataas na antas ito ay sumasaklaw sa: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- retrieval drift at sirang mga hangganan ng konteksto -- walang laman o lipas na mga index at mga tindahan ng vector -- pag-embed laban sa semantic mismatch -- agarang pagpupulong at mga isyu sa window ng konteksto -- pagbagsak ng lohika at sobrang kumpiyansa na mga sagot -- mahabang kadena at mga pagkabigo sa koordinasyon ng ahente -- multi agent memory at role drift -- mga problema sa pag-deploy at pag-order ng bootstrap +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -Ang ideya ay simple: +The idea is simple: -1. Kapag nag-imbestiga ka ng masamang tugon, kunin ang: - - gawain at kahilingan ng user - - ruta o provider combo sa OmniRoute - - anumang konteksto ng RAG na ginamit sa ibaba ng agos (mga nakuhang dokumento, mga tawag sa tool, atbp) -2. I-mapa ang insidente sa isa o dalawang numero ng WFGY ProblemMap (`No.1` … `No.16`). -3. Itago ang numero sa sarili mong dashboard, runbook, o incident tracker sa tabi ng mga log ng OmniRoute. -4. Gamitin ang kaukulang pahina ng WFGY upang magpasya kung kailangan mong baguhin ang iyong RAG stack, retriever, o diskarte sa pagruruta. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -Dito nakatira ang buong teksto at mga konkretong recipe (lisensya ng MIT, text lang): +Full text and concrete recipes live here (MIT license, text only): [WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -Maaari mong balewalain ang seksyong ito kung hindi ka nagpapatakbo ng RAG o mga pipeline ng ahente sa likod ng OmniRoute.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**Mga Isyu sa GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Arkitektura**: Tingnan ang [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) para sa mga panloob na detalye -**API Reference**: Tingnan ang [`docs/API_REFERENCE.md`](API_REFERENCE.md) para sa lahat ng endpoint -**Dashboard ng Kalusugan**: Suriin ang**Dashboard → Kalusugan**para sa real-time na status ng system -**Translator**: Gamitin ang**Dashboard → Translator**para i-debug ang mga isyu sa format +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/phi/llm.txt b/docs/i18n/phi/llm.txt new file mode 100644 index 0000000000..96c2089dd0 --- /dev/null +++ b/docs/i18n/phi/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Filipino) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Pangkalahatang-ideya + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Seguridad +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/pl/README.md b/docs/i18n/pl/README.md index ac06f50863..c86e155267 100644 --- a/docs/i18n/pl/README.md +++ b/docs/i18n/pl/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Twój uniwersalny serwer proxy API — jeden punkt końcowy, ponad 60 dostawców, zero przestojów. Teraz z**serwerem MCP (25 narzędzi)**,**protokołem A2A**,**systemami pamięci/umiejętności**i**aplikacją Electron Desktop**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Uzupełnianie czatu • Osadzanie • Generowanie obrazu • Wideo • Muzyka • Dźwięk • Zmiana rankingu •**Wyszukiwanie w Internecie**• Serwer MCP • Protokół A2A • 100% TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Twój uniwersalny serwer proxy API — jeden punkt końcowy, ponad 60 dostawcó [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Strona internetowa](https://omniroute.online) • [🚀 Szybki start](#-szybki start) • [💡 Funkcje](#-kluczowe funkcje) • [📖 Dokumenty](#-dokumentacja) • [💰 Ceny](#-ceny w skrócie) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Dostępne w:**🇺🇸 [angielski](README.md) | 🇧🇷 [Portugalski (Brazylia)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Francja](docs/i18n/fr/README.md) | 🇮🇹 [włoski](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonezja](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugalia)](docs/i18n/pt/README.md) | 🇷🇴 [Romański](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Słowenia](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [filipiński](docs/i18n/phi/README.md) | 🇨🇿 [Ceština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,554 +60,629 @@ _Twój uniwersalny serwer proxy API — jeden punkt końcowy, ponad 60 dostawcó ## 📸 Dashboard Preview - +
+Click to see dashboard screenshots -Kliknij, aby zobaczyć zrzuty ekranu panelu +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | -| Strona | Zrzut ekranu | -| ------------------------- | --------------------------------------------------- | ---------- | -| **Dostawcy** | ![Dostawcy](docs/screenshots/01-providers.png) | -| **Kombinacje** | ![Kombinacje](docs/screenshots/02-combos.png) | -| **Analiza** | ![Analytics](docs/screenshots/03-analytics.png) | -| **Zdrowie** | ![Zdrowie](docs/screenshots/04-health.png) | -| **Tłumacz** | ![Tłumacz](docs/screenshots/05-translator.png) | -| **Ustawienia** | ![Ustawienia](docs/screenshots/06-settings.png) | -| **Narzędzia CLI** | ![Narzędzia CLI](docs/screenshots/07-cli-tools.png) | -| **Dzienniki użytkowania** | ![Wykorzystanie](docs/screenshots/08-usage.png) | -| **Punkty końcowe** | ![Punkty końcowe](docs/screenshots/09-endpoint.png) |
| + --- ### 🤖 Free AI Provider for your favorite coding agents -_Połącz dowolne narzędzie IDE lub CLI oparte na sztucznej inteligencji poprzez OmniRoute — bezpłatną bramę API dla nieograniczonego kodowania._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ - + - - - - - - - - - - - +
+ - OpenClaw
+ OpenClaw
OpenClaw

- ⭐ 205 tys. + ⭐ 205K
+ - NanoBot
+ NanoBot
NanoBot

- ⭐ 20,9 tys. + ⭐ 20.9K
+ - PicoClaw
+ PicoClaw
PicoClaw

- ⭐ 14,6 tys. + ⭐ 14.6K
+ - ZeroClaw
+ ZeroClaw
ZeroClaw

- ⭐ 9,9 tys. + ⭐ 9.9K
+ - IronClaw
- Żelazny Pazur + IronClaw
+ IronClaw

- ⭐ 2,1 tys. + ⭐ 2.1K
+ - OpenCode
+ OpenCode
OpenCode

- ⭐ 106 tys. + ⭐ 106K
+ - Codex CLI
- CLI Kodeksu + Codex CLI
+ Codex CLI

- ⭐ 60,8 tys. + ⭐ 60.8K
+ - Kod Claude'a
+ Claude Code
Claude Code

- ⭐ 67,3 tys. + ⭐ 67.3K
+ - Gemini CLI
+ Gemini CLI
Gemini CLI

- ⭐ 94,7 tys. + ⭐ 94.7K
+ - Kod Kilo
- Kod kilo + Kilo Code
+ Kilo Code

- ⭐ 15,5 tys. + ⭐ 15.5K
-📡 Wszyscy agenci łączą się przez http://localhost:20128/v1 lub http://cloud.omniroute.online/v1 — jedna konfiguracja, nieograniczona liczba modeli i przydział--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Przestań marnować pieniądze i przekraczać limity:** +**Stop wasting money and hitting limits:** -- Limit subskrypcji wygasa niewykorzystany co miesiąc -- Limity szybkości zatrzymują Cię w połowie kodowania -- Drogie interfejsy API (20-50 USD miesięcznie na dostawcę) -- Ręczne przełączanie pomiędzy dostawcami +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute rozwiązuje ten problem:** +**OmniRoute solves this:** -- ✅**Maksymalizuj liczbę subskrypcji**- Śledź limit, wykorzystaj każdy bit przed zresetowaniem -- ✅**Automatyczny powrót**- Subskrypcja → Klucz API → Tani → Bezpłatny, zero przestojów -- ✅**Wiele kont**- Praca okrężna pomiędzy kontami każdego dostawcy -- ✅**Uniwersalny**- Działa z Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw i dowolnym narzędziem CLI--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Dołącz do naszej społeczności!**[Grupa WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Uzyskaj pomoc, dziel się wskazówkami i bądź na bieżąco. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Strona internetowa**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Problemy**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Grupa społeczności](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Wspieranie**: zobacz [CONTRIBUTING.md](CONTRIBUTING.md), otwórz PR lub wybierz „dobry pierwszy numer” -**Oryginalny projekt**: [9router autorstwa Decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Otwierając zgłoszenie, uruchom komendę system-info i załącz wygenerowany plik:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Spowoduje to wygenerowanie pliku `system-info.txt` z wersją Node.js, wersją OmniRoute, szczegółami systemu operacyjnego, zainstalowanymi narzędziami CLI (qoder, gemini, claude, codex, antygrawitacja, droid itp.), statusem Docker/PM2 i pakietami systemowymi — wszystko, czego potrzebujemy, aby szybko odtworzyć problem. Dołącz plik bezpośrednio do swojego zgłoszenia w GitHubie.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Każdy programista korzystający z narzędzi AI codziennie spotyka się z tymi problemami.**OmniRoute został stworzony, aby rozwiązać je wszystkie — od przekroczeń kosztów po blokady regionalne, od zepsutych przepływów OAuth po operacje protokołów i obserwowalność przedsiębiorstwa. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. - -💸 1. „Płacę za drogi abonament, a mimo to przeszkadzają mi limity” +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Programiści płacą 20–200 USD miesięcznie za Claude Pro, Codex Pro lub GitHub Copilot. Nawet płacąc, limit ma pułap – 5 godzin użytkowania, limity tygodniowe lub limity stawek za minutę. W połowie sesji kodowania dostawca przestaje odpowiadać, a programista traci płynność i produktywność. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Jak rozwiązuje to OmniRoute:** +**How OmniRoute solves it:** --**Inteligentny 4-poziomowy powrót**— Jeśli limit subskrypcji się wyczerpie, automatycznie przekierowuje do klucza API → Tani → Bezpłatny bez ręcznej interwencji --**Śledzenie limitów dostawcy**— Odświeżanie migawek przydziałów w pamięci podręcznej zgodnie z harmonogramem po stronie serwera (domyślnie `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) z możliwością ręcznego odświeżania w interfejsie użytkownika --**Obsługa wielu kont**— Wiele kont na dostawcę z funkcją automatycznego przełączania między kontami — gdy skończy się jedno, następuje przejście do następnego --**Niestandardowe kombinacje**— Konfigurowalne łańcuchy rezerwowe z 9 strategiami równoważenia (priorytet, ważone, pierwsze wypełnienie, działanie okrężne, P2C, losowe, najrzadziej używane, zoptymalizowane pod względem kosztów, ściśle losowe) --**Przydziały biznesowe Kodeksu**— monitorowanie przydziałów przestrzeni roboczej firmy/zespołu bezpośrednio w panelu kontrolnym
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard - -🔌 2. „Muszę korzystać z wielu dostawców, ale każdy ma inny interfejs API” + -OpenAI używa jednego formatu, Claude (Anthropic) używa innego, Gemini jeszcze innego. Jeśli programista chce przetestować modele od różnych dostawców lub korzystać z nich w trybie awaryjnym, musi ponownie skonfigurować pakiety SDK, zmienić punkty końcowe i poradzić sobie z niekompatybilnymi formatami. Dostawcy niestandardowi (FriendLI, NIM) mają niestandardowe punkty końcowe modelu. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Jak rozwiązuje to OmniRoute:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Ujednolicony punkt końcowy**— pojedynczy adres „http://localhost:20128/v1” służy jako serwer proxy dla wszystkich ponad 60 dostawców --**Tłumaczenie formatu**— Automatyczne i przejrzyste: OpenAI ↔ Claude ↔ Gemini ↔ API odpowiedzi --**Oczyszczanie odpowiedzi**— Usuwa niestandardowe pola (`x_groq`, `usage_breakdown`, `service_tier`), które psują OpenAI SDK v1.83+ --**Normalizacja ról**— Konwertuje „programista” → „system” dla dostawców innych niż OpenAI; `system` → `użytkownik` dla GLM/ERNIE --**Wyodrębnianie tagów Think**— wyodrębnia bloki „” z modeli takich jak DeepSeek R1 do standardowej „treści_reasoning_content” --**Strukturalne dane wyjściowe dla Gemini**— automatyczna konwersja `json_schema` → `responseMimeType`/`responseSchema` --**`stream` domyślnie ma wartość `false`**— Zgodność ze specyfikacją OpenAI, unikanie nieoczekiwanego SSE w pakietach SDK Python/Rust/Go
+**How OmniRoute solves it:** - -🌐 3. „Mój dostawca AI blokuje mój region/kraj” +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Dostawcy tacy jak OpenAI/Codex blokują dostęp z określonych regionów geograficznych. Podczas połączeń OAuth i API użytkownicy otrzymują błędy typu „unsupported_country_region_territory”. Jest to szczególnie frustrujące dla programistów z krajów rozwijających się. + -**Jak rozwiązuje to OmniRoute:** +
+🌐 3. "My AI provider blocks my region/country" --**3-poziomowa konfiguracja serwera proxy**— Konfigurowalny serwer proxy na 3 poziomach: globalny (cały ruch), na dostawcę (tylko jeden dostawca) i na połączenie/klucz --**Oznaczone kolorami identyfikatory proxy**— Wskaźniki wizualne: 🟢 globalny serwer proxy, 🟡 serwer proxy dostawcy, 🔵 serwer proxy połączenia, zawsze pokazujący adres IP --**Wymiana tokenów OAuth przez serwer proxy**— przepływ OAuth przechodzi również przez serwer proxy, rozwiązując problem „unsupported_country_region_territory” --**Test połączenia przez serwer proxy**— Testy połączenia wykorzystują skonfigurowany serwer proxy (koniec z bezpośrednim obejściem) --**Obsługa SOCKS5**— Pełna obsługa proxy SOCKS5 dla routingu wychodzącego --**Podrabianie odcisków palców TLS**— Odcisk palca TLS podobny do przeglądarki za pośrednictwem `wreq-js` w celu ominięcia wykrywania botów --**🔏 Dopasowanie odcisków palców CLI**— zmienia kolejność nagłówków i pól treści, aby dopasować je do natywnych podpisów binarnych CLI, drastycznie zmniejszając ryzyko oznaczania konta. Adres IP serwera proxy zostaje zachowany — jednocześnie uzyskujesz maskowanie IP**i**IP
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. - -🆓 4. „Chcę używać sztucznej inteligencji do kodowania, ale nie mam pieniędzy” +**How OmniRoute solves it:** -Nie każdy może zapłacić 20–200 USD miesięcznie za subskrypcje AI. Studenci, programiści z krajów wschodzących, hobbyści i freelancerzy potrzebują dostępu do wysokiej jakości modeli po zerowych kosztach. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Jak rozwiązuje to OmniRoute:** + --**Wbudowani dostawcy Free Tier**— Natywne wsparcie dla w 100% darmowych dostawców: Qoder (5 nieograniczonych modeli przez OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 nieograniczone modele: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, Vision-model), Kiro (Claude + AWS Builder ID za darmo), Gemini CLI (180 tys. tokenów miesięcznie za darmo) --**Ollama Cloud**— modele Ollama hostowane w chmurze na `api.ollama.com` z bezpłatnym poziomem „Lekkie użycie”; użyj przedrostka `ollamacloud/` --**Kombinacje dostępne tylko bezpłatnie**— Łańcuch `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = 0 USD/miesiąc bez przestojów --**Bezpłatny dostęp do NVIDIA NIM**— bezpłatny dostęp dla deweloperów przy ~40 obr./min na zawsze do ponad 70 modeli na stronie build.nvidia.com (przejście z kredytów na same limity szybkości) --**Strategia zoptymalizowana pod względem kosztów**— Strategia routingu, która automatycznie wybiera najtańszego dostępnego dostawcę +
+🆓 4. "I want to use AI for coding but I have no money" - -🔒 5. „Muszę chronić moją bramę AI przed nieautoryzowanym dostępem” +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -Podczas udostępniania bramy AI w sieci (LAN, VPS, Docker) każda osoba posiadająca adres może wykorzystać tokeny/przydział programisty. Bez ochrony interfejsy API są podatne na niewłaściwe użycie, natychmiastowe wstrzyknięcie i nadużycia. +**How OmniRoute solves it:** -**Jak rozwiązuje to OmniRoute:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**Zarządzanie kluczami API**— generowanie, rotacja i określanie zakresu dla każdego dostawcy z dedykowaną stroną `/dashboard/api-manager` --**Uprawnienia na poziomie modelu**— Ogranicz klucze API do określonych modeli (`openai/*`, wzorce symboli wieloznacznych), z przełącznikiem Zezwalaj na wszystko/Ogranicz --**API Endpoint Protection**— Wymagaj klucza dla `/v1/models` i blokuj określonych dostawców na liście --**Auth Guard + ochrona CSRF**— Wszystkie trasy dashboardu chronione oprogramowaniem pośredniczącym `withAuth` + tokenami CSRF --**Rate Limiter**— Ograniczanie szybkości na IP z konfigurowalnymi oknami --**Filtrowanie IP**— Lista dozwolonych/blokowanych do kontroli dostępu --**Szybka ochrona przed wstrzyknięciem**— Oczyszczanie przed złośliwymi wzorcami podpowiedzi --**Szyfrowanie AES-256-GCM**— Poświadczenia szyfrowane w stanie spoczynku
+ - -🛑 6. „Mój dostawca przestał działać i straciłem płynność kodowania” +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -Dostawcy sztucznej inteligencji mogą stać się niestabilni, zwracać błędy 5xx lub przekraczać tymczasowe limity szybkości. Jeśli programista jest zależny od jednego dostawcy, jego praca jest przerywana. Bez wyłączników automatycznych wielokrotne próby mogą spowodować awarię aplikacji. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Jak rozwiązuje to OmniRoute:** +**How OmniRoute solves it:** --**Wyłącznik w zależności od modelu**— Automatyczne otwieranie/zamykanie z konfigurowalnymi progami i czasem schładzania (zamknięty/otwarty/półotwarty), zakres zależny od modelu, aby uniknąć blokad kaskadowych --**Wykładniczy wycofywanie**— Stopniowe opóźnienia ponownych prób --**Anti-Thundering Herd**— Mutex + ochrona semaforów przed równoczesnymi burzami ponownych prób --**Łańcuchy awaryjne typu Combo**— jeśli główny dostawca zawiedzie, automatycznie przejdzie przez łańcuch bez interwencji --**Wyłącznik automatyczny**— automatycznie wyłącza niesprawnych dostawców w łańcuchu combo --**Panel stanu**— Monitorowanie czasu pracy, stany wyłączników, blokady, statystyki pamięci podręcznej, opóźnienia p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest - -🔧 7. „Konfigurowanie każdego narzędzia AI jest żmudne i powtarzalne” + -Programiści używają Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Każde narzędzie wymaga innej konfiguracji (punkt końcowy API, klucz, model). Ponowna konfiguracja w przypadku zmiany dostawcy lub modelu jest stratą czasu. +
+🛑 6. "My provider went down and I lost my coding flow" -**Jak rozwiązuje to OmniRoute:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**Panel narzędzi CLI**— Dedykowana strona z konfiguracją jednym kliknięciem dla Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline --**Generator konfiguracji GitHub Copilot**— Generuje plik „chatLanguageModels.json” dla kodu VS z zbiorczym wyborem modelu --**Kreator wprowadzenia**— konfiguracja w 4 krokach dla początkujących użytkowników --**Jeden punkt końcowy, wszystkie modele**— Skonfiguruj raz `http://localhost:20128/v1`, uzyskaj dostęp do ponad 60 dostawców
+**How OmniRoute solves it:** - -🔑 8. „Zarządzanie tokenami OAuth od wielu dostawców to piekło” +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot — wszystkie korzystają z OAuth 2.0 z wygasającymi tokenami. Programiści muszą stale przeprowadzać ponowne uwierzytelnianie, radzić sobie z „brakującym sekretu_klienta”, „przekierowaniem_uri_mismatch” i awariami na zdalnych serwerach. Szczególnie problematyczny jest protokół OAuth w sieci LAN/VPS. + -**Jak rozwiązuje to OmniRoute:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Automatyczne odświeżanie tokenu**— tokeny OAuth odświeżają się w tle przed wygaśnięciem --**Wbudowany OAuth 2.0 (PKCE)**— Automatyczny przepływ dla Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder --**Wielokontowy OAuth**— wiele kont na dostawcę poprzez ekstrakcję tokenów JWT/ID --**OAuth LAN/Poprawka zdalna**— Wykrywanie prywatnego adresu IP dla `redirect_uri` + ręczny tryb adresu URL dla serwerów zdalnych --**OAuth Behind Nginx**— Używa `window.location.origin` w celu zapewnienia zgodności z odwrotnym proxy --**Przewodnik po zdalnym OAuth**— szczegółowy przewodnik dotyczący danych uwierzytelniających Google Cloud na platformie VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. - -📊 9. „Nie wiem, ile i gdzie wydaję” +**How OmniRoute solves it:** -Programiści korzystają z wielu płatnych dostawców, ale nie mają jednolitego widoku wydatków. Każdy dostawca ma własny pulpit rozliczeniowy, ale nie ma widoku skonsolidowanego. Nieoczekiwane koszty mogą się kumulować. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Jak rozwiązuje to OmniRoute:** + --**Panel analizy kosztów**— śledzenie kosztów według tokenu i zarządzanie budżetem dla każdego dostawcy --**Limity budżetowe na poziom**— Pułap wydatków na poziom, który uruchamia automatyczne wycofanie --**Konfiguracja cen dla poszczególnych modeli**— Ceny dla poszczególnych modeli można konfigurować --**Statystyki użytkowania na klucz API**— Liczba żądań i znacznik czasu ostatniego użycia na klucz --**Panel analityczny**— karty statystyk, wykres wykorzystania modelu, tabela dostawców ze wskaźnikami powodzenia i opóźnieniami +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" - -🐛 10. „Nie mogę diagnozować błędów i problemów w wywołaniach AI” +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Gdy połączenie nie powiedzie się, programista nie wie, czy był to limit szybkości, wygasły token, nieprawidłowy format czy błąd dostawcy. Fragmentaryczne dzienniki na różnych terminalach. Bez obserwowalności debugowanie odbywa się metodą prób i błędów. +**How OmniRoute solves it:** -**Jak rozwiązuje to OmniRoute:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Ujednolicony pulpit nawigacyjny**— 4 karty: Dzienniki żądań, Dzienniki proxy, Dzienniki audytu, Konsola --**Przeglądarka logów w konsoli**— Przeglądarka działająca w stylu terminala w czasie rzeczywistym z poziomami oznaczonymi kolorami, automatycznym przewijaniem, wyszukiwaniem i filtrowaniem --**Dzienniki proxy SQLite**— trwałe dzienniki, które przetrwają ponowne uruchomienie serwera --**Plac zabaw dla tłumaczy**— 4 tryby debugowania: Plac zabaw (tłumaczenie formatu), Tester czatu (w obie strony), Stanowisko testowe (wsadowe), Monitor na żywo (w czasie rzeczywistym) --**Żądanie telemetrii**— opóźnienie p50/p95/p99 + śledzenie identyfikatora X-Request-Id --**Logowanie oparte na plikach z rotacją**— dzienniki aplikacji zmieniają się według rozmiaru, dni przechowywania i liczby archiwów; Artefakty dziennika połączeń zmieniają się według dni przechowywania i liczby plików --**Raport informacji o systemie**— `npm run system-info` generuje `system-info.txt` z pełnym środowiskiem (wersja węzła, wersja OmniRoute, system operacyjny, narzędzia CLI, status Docker/PM2). Dołącz go podczas zgłaszania problemów w celu natychmiastowej selekcji.
+ - -🏗️ 11. „Wdrażanie i konserwacja bramy jest skomplikowane” +
+📊 9. "I don't know how much I'm spending or where" -Instalacja, konfiguracja i utrzymanie serwera proxy AI w różnych środowiskach (lokalnym, VPS, Docker, chmura) jest pracochłonne. Problemy takie jak zakodowane na stałe ścieżki, „EACCES” w katalogach, konflikty portów i kompilacje międzyplatformowe zwiększają tarcia. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Jak rozwiązuje to OmniRoute:** +**How OmniRoute solves it:** --**npm globalna instalacja**— `npm install -g omniroute && omniroute` — gotowe --**Docker Multi-platform**— natywny AMD64 + ARM64 (Apple Silicon, AWS Graviton, Raspberry Pi) --**Profile Docker Compose**— `base` (bez narzędzi CLI) i `cli` (z Claude Code, Codex, OpenClaw) --**Electron Desktop App**— Natywna aplikacja dla systemów Windows/macOS/Linux z zasobnikiem systemowym, automatycznym uruchamianiem i trybem offline --**Tryb Split-Port**— API i pulpit nawigacyjny na oddzielnych portach dla zaawansowanych scenariuszy (odwrotne proxy, sieć kontenerowa) --**Cloud Sync**— skonfiguruj synchronizację między urządzeniami za pośrednictwem Cloudflare Workers --**DB Backup**— Automatyczne tworzenie kopii zapasowych, przywracanie, eksportowanie i importowanie wszystkich ustawień z opcją „DISABLE_SQLITE_AUTO_BACKUP” dla kopii zapasowych zarządzanych zewnętrznie
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency - -🌍 12. „Interfejs jest wyłącznie w języku angielskim, a mój zespół nie mówi po angielsku” + -Zespoły w krajach nieanglojęzycznych, szczególnie w Ameryce Łacińskiej, Azji i Europie, mają trudności z interfejsami dostępnymi wyłącznie w języku angielskim. Bariery językowe ograniczają wdrażanie i zwiększają liczbę błędów konfiguracyjnych. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Jak rozwiązuje to OmniRoute:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Panel i18n — 30 języków**— Przetłumaczono ponad 500 klawiszy, w tym arabski, bułgarski, duński, niemiecki, hiszpański, fiński, francuski, hebrajski, hindi, węgierski, indonezyjski, włoski, japoński, koreański, malajski, holenderski, norweski, polski, portugalski (PT/BR), rumuński, rosyjski, słowacki, szwedzki, tajski, ukraiński, wietnamski, chiński, filipiński, angielski --**Obsługa RTL**— obsługa tekstu od prawej do lewej w języku arabskim i hebrajskim --**Wielojęzyczne pliki README**— 30 kompletnych tłumaczeń dokumentacji --**Wybór języka**— Ikona kuli ziemskiej w nagłówku umożliwiająca przełączanie w czasie rzeczywistym
+**How OmniRoute solves it:** - -🔄 13. „Potrzebuję czegoś więcej niż czatu — potrzebuję osadzania, obrazów i dźwięku” +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -Sztuczna inteligencja to nie tylko ukończenie czatu. Twórcy muszą generować obrazy, transkrybować dźwięk, tworzyć osadzania dla RAG, zmieniać rangę dokumentów i moderować treści. Każdy interfejs API ma inny punkt końcowy i format. + -**Jak rozwiązuje to OmniRoute:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Osadzania**— `/v1/osadzania` u 6 dostawców i ponad 9 modeli --**Generowanie obrazów**— `/v1/images/generacje` z 10 dostawcami i ponad 20 modelami (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Tekst na wideo**— `/v1/videos/generacje` — ComfyUI (AnimateDiff, SVD) i SD WebUI --**Tekst na muzykę**— `/v1/muzyka/generacje` — ComfyUI (Stable Audio Open, MusicGen) --**Transkrypcja audio**— `/v1/audio/transkrypcje` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Zamiana tekstu na mowę**— `/v1/audio/mowa` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + obecni dostawcy --**Moderacje**— `/v1/moderacje` — Sprawdzanie bezpieczeństwa treści --**Reranking**— `/v1/rerank` — Zmiana rankingu trafności dokumentu --**Responses API**— Pełna obsługa `/v1/responses` dla Codexu
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. - -🧪 14. „Nie mam możliwości przetestowania i porównania jakości różnych modeli” +**How OmniRoute solves it:** -Programiści chcą wiedzieć, który model jest najlepszy dla ich przypadku użycia — kodu, tłumaczenia, rozumowania — ale ręczne porównywanie jest powolne. Nie istnieją żadne zintegrowane narzędzia eval. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**Jak rozwiązuje to OmniRoute:** + --**Oceny LLM**— Testowanie złotego zestawu z 10 fabrycznie załadowanymi skrzynkami obejmującymi pozdrowienia, matematykę, geografię, generowanie kodu, zgodność z JSON, tłumaczenie, przeceny, odmowy ze względów bezpieczeństwa --**4 strategie dopasowania**— „dokładne”, „zawiera”, „regex”, „niestandardowe” (funkcja JS) --**Stolik testowy dla tłumaczy**— Testowanie wsadowe z wieloma danymi wejściowymi i oczekiwanymi wynikami, porównanie między dostawcami --**Tester czatu**— Pełna podróż w obie strony z renderowaniem odpowiedzi wizualnych --**Live Monitor**— Strumień w czasie rzeczywistym wszystkich żądań przepływających przez serwer proxy +
+🌍 12. "The interface is English-only and my team doesn't speak English" - -📈 15. „Muszę skalować bez utraty wydajności” +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -W miarę wzrostu liczby żądań bez buforowania tych samych pytań generowane są podwójne koszty. Bez idempotencji zduplikowane żądania przetwarzania odpadów. Należy przestrzegać limitów stawek dla poszczególnych dostawców. +**How OmniRoute solves it:** -**Jak rozwiązuje to OmniRoute:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Semantyczna pamięć podręczna**— Dwuwarstwowa pamięć podręczna (podpis + semantyka) zmniejsza koszty i opóźnienia --**Request Idempotency**— okno deduplikacji 5 s dla identycznych żądań --**Wykrywanie limitów szybkości**— RPM na dostawcę, minimalna przerwa i maksymalne jednoczesne śledzenie --**Edytowalne limity szybkości**— Konfigurowalne ustawienia domyślne w Ustawieniach → Odporność z trwałością --**API Key Validation Cache**— 3-warstwowa pamięć podręczna zapewniająca wydajność produkcyjną --**Panel kontrolny stanu z telemetrią**— opóźnienia p50/p95/p99, statystyki pamięci podręcznej, czas pracy
+ - -🤖 16. „Chcę kontrolować zachowanie modelu globalnie” +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Deweloperzy, którzy chcą wszystkich odpowiedzi w określonym języku, w określonym tonie lub chcą ograniczyć tokeny rozumowania. Konfigurowanie tego w każdym narzędziu/żądaniu jest niepraktyczne. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Jak rozwiązuje to OmniRoute:** +**How OmniRoute solves it:** --**Wstrzykiwanie monitu systemowego**— monit globalny stosowany do wszystkich żądań --**Przemyślana weryfikacja budżetu**— Kontrola alokacji tokenów rozumowania na każde żądanie (przejściowe, automatyczne, niestandardowe, adaptacyjne) --**9 Strategii routingu**— Globalne strategie określające sposób dystrybucji żądań --**Wildcard Router**— wzorce `provider/*` kierują dynamicznie do dowolnego dostawcy --**Przełączanie kombinacji włącz/wyłącz**— przełączaj kombinacje bezpośrednio z pulpitu nawigacyjnego --**Przełączanie dostawcy**— Włącz/wyłącz wszystkie połączenia dla dostawcy jednym kliknięciem --**Zablokowani dostawcy**— Wyklucz określonych dostawców z listy `/v1/models`
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex - -🧰 17. „Potrzebuję narzędzi MCP jako produktów najwyższej klasy” + -Wiele bram AI ujawnia MCP jedynie jako ukryty szczegół implementacji. Zespoły potrzebują widocznej, zarządzalnej warstwy operacyjnej. +
+🧪 14. "I have no way to test and compare quality across models" -**Jak rozwiązuje to OmniRoute:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP pojawia się w panelu nawigacji na desce rozdzielczej i w zakładce protokołu punktu końcowego -- Dedykowana strona zarządzania MCP z procesem, narzędziami, zakresami i audytem -- Wbudowany szybki start dla `omniroute --mcp` i dołączania klientów
+**How OmniRoute solves it:** - -🧠 18. „Potrzebuję orkiestracji A2A ze ścieżkami zadań synchronizacji i strumieniowania” +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Przepływy pracy agentów wymagają zarówno bezpośrednich odpowiedzi, jak i długotrwałego wykonywania strumieniowego z kontrolą cyklu życia. + -**Jak rozwiązuje to OmniRoute:** +
+📈 15. "I need to scale without losing performance" -- Punkt końcowy A2A JSON-RPC („POST /a2a”) z „wiadomością/wyślij” i „wiadomością/strumień” -- Przesyłanie strumieniowe SSE z propagacją stanu terminala -- Interfejsy API cyklu życia zadań dla „zadań/pobierz” i „zadań/anuluj”.
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. - -🛰️ 19. „Potrzebuję prawdziwego stanu procesu MCP, a nie zgadywanego statusu” +**How OmniRoute solves it:** -Zespoły operacyjne muszą wiedzieć, czy MCP rzeczywiście żyje, a nie tylko, czy można uzyskać dostęp do interfejsu API. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**Jak rozwiązuje to OmniRoute:** + -- Plik pulsu środowiska wykonawczego z PID, znacznikami czasu, transportem, liczbą narzędzi i trybem zakresu -- API statusu MCP łączące puls + ostatnią aktywność -- Karty stanu interfejsu użytkownika dotyczące świeżości procesów/czasu pracy/bicia serca +
+🤖 16. "I want to control model behavior globally" - -📋 20. „Potrzebuję wykonania narzędzia MCP z możliwością audytu” +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Gdy narzędzia modyfikują konfigurację lub uruchamiają działania operacyjne, zespoły potrzebują identyfikowalności kryminalistycznej. +**How OmniRoute solves it:** -**Jak rozwiązuje to OmniRoute:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- Rejestrowanie audytu wspierane przez SQLite dla wywołań narzędzi MCP -- Filtruje według narzędzia, sukcesu/porażki, klucza API i paginacji -- Tabela audytu pulpitu nawigacyjnego + punkty końcowe statystyk dla automatyzacji
+ - -🔐 21. „Potrzebuję uprawnień MCP o określonym zakresie na integrację” +
+🧰 17. "I need MCP tools as first-class product capabilities" -Różni klienci powinni mieć najniższy dostęp do kategorii narzędzi. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Jak rozwiązuje to OmniRoute:** +**How OmniRoute solves it:** -- 10 szczegółowych zakresów MCP zapewniających kontrolowany dostęp do narzędzi -- Egzekwowanie zakresu i widoczność w interfejsie zarządzania MCP -- Bezpieczna domyślna pozycja dla narzędzi operacyjnych
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding - -⚙️ 22. „Potrzebuję kontroli operacyjnej bez konieczności ponownego wdrażania” + -Zespoły potrzebują szybkich zmian w czasie działania podczas incydentów lub zdarzeń kosztowych. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**Jak rozwiązuje to OmniRoute:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Aktywacja kombinacji przełączników bezpośrednio z pulpitu nawigacyjnego MCP -- Zastosuj profile odporności ze wstępnie zdefiniowanych pakietów zasad -- Zresetuj stan wyłącznika automatycznego z tego samego panelu operacyjnego
+**How OmniRoute solves it:** - -🔄 23. „Potrzebuję widoczności i anulowania cyklu życia zadań A2A na żywo” +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Bez widoczności cyklu życia zdarzenia związane z zadaniami stają się trudne do segregacji. + -**Jak rozwiązuje to OmniRoute:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Lista zadań/filtrowanie według stanu/umiejętności z paginacją -- Szczegółowa analiza metadanych zadań, zdarzeń i artefaktów -- Punkt końcowy anulowania zadania i akcja interfejsu użytkownika z potwierdzeniem
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. - -🌊 24. „Potrzebuję wskaźników aktywnego strumienia dla obciążenia A2A” +**How OmniRoute solves it:** -Przepływy pracy związane z przesyłaniem strumieniowym wymagają operacyjnego wglądu w współbieżność i połączenia na żywo. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**Jak rozwiązuje to OmniRoute:** + -- Aktywne liczniki strumieni zintegrowane ze statusem A2A -- Znacznik czasu ostatniego zadania i liczba stanów -- Karty pulpitu A2A do monitorowania operacji w czasie rzeczywistym +
+📋 20. "I need auditable MCP tool execution" - -🪪 25. „Potrzebuję standardowego wyszukiwania agentów dla klientów” +When tools mutate config or trigger ops actions, teams need forensic traceability. -Zewnętrzni klienci i koordynatorzy potrzebują metadanych do odczytu maszynowego na potrzeby wdrożenia. +**How OmniRoute solves it:** -**Jak rozwiązuje to OmniRoute:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- Karta agenta ujawniona w `/.well-known/agent.json` -- Możliwości i umiejętności pokazane w interfejsie zarządzania -- Interfejs API stanu A2A zawiera metadane wykrywania do automatyzacji
+ - -🧭 26. „Potrzebuję wykrywalności protokołu w UX produktu” +
+🔐 21. "I need scoped MCP permissions per integration" -Jeśli użytkownicy nie mogą odkryć powierzchni protokołu, spada jakość przyjęcia i wsparcia. +Different clients should have least-privilege access to tool categories. -**Jak rozwiązuje to OmniRoute:** +**How OmniRoute solves it:** -- Skonsolidowana strona**Punkty końcowe**z zakładkami dla punktów końcowych proxy, MCP, A2A i API -- Przełączniki stanu usługi Inline (Online/Offline) dla MCP i A2A -- Linki z przeglądu do dedykowanych zakładek zarządzania
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling - -🧪 27. „Potrzebuję kompleksowej weryfikacji protokołu z prawdziwymi klientami” + -Testy próbne nie wystarczą do sprawdzenia zgodności protokołu przed wydaniem. +
+⚙️ 22. "I need operational controls without redeploying" -**Jak rozwiązuje to OmniRoute:** +Teams need quick runtime changes during incidents or cost events. -- Pakiet E2E, który uruchamia aplikację i wykorzystuje prawdziwy transport klienta MCP SDK -- Testy klienta A2A pod kątem wykrywania, wysyłania, przesyłania strumieniowego, pobierania i anulowania przepływów -- Sprawdzaj twierdzenia względem interfejsów API audytu MCP i zadań A2A
+**How OmniRoute solves it:** - -📡 28. „Potrzebuję ujednoliconej obserwowalności we wszystkich interfejsach” +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -Podział obserwowalności według protokołu tworzy martwe punkty i wydłuża MTTR. + -**Jak rozwiązuje to OmniRoute:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Ujednolicone dashboardy/dzienniki/analizy w jednym produkcie -- Stan + audyt + telemetria żądań w warstwach OpenAI, MCP i A2A -- Operacyjne interfejsy API dla statusu i automatyzacji
+Without lifecycle visibility, task incidents become hard to triage. - -💼 29. „Potrzebuję jednego środowiska wykonawczego dla proxy + narzędzi + orkiestracji agentów” +**How OmniRoute solves it:** -Uruchamianie wielu oddzielnych usług zwiększa koszty operacyjne i tryby awarii. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**Jak rozwiązuje to OmniRoute:** + -- Serwer proxy zgodny z OpenAI, serwer MCP i serwer A2A w jednym stosie -- Wspólne uwierzytelnianie, odporność, magazyn danych i obserwowalność -- Spójny model polityki na wszystkich płaszczyznach interakcji +
+🌊 24. "I need active stream metrics for A2A load" - -🚀 30. „Muszę dostarczać agentowe przepływy pracy bez konieczności ciągłego tworzenia kodu klejącego” +Streaming workflows require operational insight into concurrency and live connections. -Zespoły tracą prędkość podczas łączenia wielu usług i skryptów ad hoc. +**How OmniRoute solves it:** -**Jak rozwiązuje to OmniRoute:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Ujednolicona strategia dotycząca punktów końcowych dla klientów i agentów -- Wbudowane interfejsy zarządzania protokołami i ścieżki sprawdzania dymu -- Podstawy gotowe do produkcji (bezpieczeństwo, logowanie, odporność, kopie zapasowe)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Poradnik A: maksymalizuj płatną subskrypcję + tanią kopię zapasową**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -608,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Poradnik B: stos kodowania o zerowym koszcie**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Poradnik C: łańcuch awaryjny działający 24 godziny na dobę, 7 dni w tygodniu**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -631,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Poradnik D: Operacje agenta z MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Skonfiguruj kodowanie AI w ciągu kilku minut w cenie**0 USD/miesiąc**. Połącz te bezpłatne konta i korzystaj z wbudowanej kombinacji**Free Stack**. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Krok | Akcja | Dostawcy odblokowani | +| Step | Action | Providers Unlocked | | ---- | -------------------------------------------------- | ------------------------------------------------------------------ | -| 1 | Połącz**Kiro**(identyfikator AWS Builder OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**bez ograniczeń**| -| 2 | Połącz**Qoder**(Google OAuth) | myślenie kimi-k2, qwen3-coder-plus, deepseek-r1... —**bez ograniczeń**| -| 3 | Połącz**Qwen**(kod urządzenia) | qwen3-coder-plus, qwen3-coder-flash... —**bez ograniczeń**| -| 4 | Połącz**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180 tys./mc za darmo**| -| 5 | `/dashboard/combos` →**Darmowy stos (0 $)**szablon | Automatycznie okrężnie wszystkich bezpłatnych dostawców | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Wskaż dowolne IDE/CLI na:**`http://localhost:20128/v1` · Klucz API: `any-string` · Gotowe. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Opcjonalna dodatkowa ochrona (również bezpłatna):**Klucz API Groq (30 obr./min za darmo), NVIDIA NIM (40 obr./min za darmo, ponad 70 modeli), Cerebras (1 mln tok/dzień), Klucz API LongCat (50 mln tokenów dziennie!), Cloudflare Workers AI (10 tys. neuronów/dzień, ponad 50 modeli).## Szybki start +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Szybki start ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **Użytkownicy pnpm:**Uruchom `pnpm Apply-builds -g` po instalacji, aby włączyć natywne skrypty kompilacji wymagane przez `better-sqlite3` i `@swc/core`: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > -> ```bzdura +> ```bash > pnpm install -g omniroute -> pnpm commit-builds -g # Wybierz wszystkie pakiety → zatwierdź +> pnpm approve-builds -g # Select all packages → approve > omniroute > ``` -Panel otwiera się pod adresem `http://localhost:20128`, a podstawowy adres URL interfejsu API to `http://localhost:20128/v1`. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Polecenie | Opis | -| ----------------------- | ------------------------------------------------------------------ | -| `omniroute` | Uruchom serwer (`PORT=20128`, API i dashboard na tym samym porcie) | -| `omniroute --port 3000` | Ustaw port kanoniczny/API na 3000 | -| `omniroute --mcp` | Uruchom serwer MCP (transport stdio) | -| `omniroute --no-open` | Nie otwieraj automatycznie przeglądarki | -| `omniroute --pomoc` | Pokaż pomoc | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Opcjonalny tryb podzielonego portu:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -W przypadku większości wdrożeń potrzebujesz tylko: +For most deployments, you only need: -| Zmienna | Domyślne | Cel | -| ------------------------ | ------------------------------ | ---------------------------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600000` | Wspólna linia bazowa dla pobierania w górę, ukryte limity czasu Undici, żądania odcisków palców TLS i limity czasu żądań mostu API/przekroczenia limitu czasu proxy | -| `STREAM_IDLE_TIMEOUT_MS` | dziedziczy `REQUEST_TIMEOUT_MS` | Maksymalna przerwa między fragmentami przesyłania strumieniowego, zanim OmniRoute przerwie strumień SSE | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -Zachowana jest kompatybilność wsteczna: istniejące `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` i inne zmienne limitu czasu dla każdej warstwy nadal działają i zastępują współdzieloną linię bazową. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Jeśli potrzebujesz lepszej kontroli, dostępne są zaawansowane ustawienia:| Zmienna | Domyślne | Cel | -| ---------------------------------------- | ------------------------------------------ | ---------------------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | dziedziczy `REQUEST_TIMEOUT_MS` | Całkowity limit czasu żądania przesyłania danych wykorzystany przez główny sygnał przerwania pobierania | -| `FETCH_HEADERS_TIMEOUT_MS` | dziedziczy `FETCH_TIMEOUT_MS` | Limit czasu Undici na otrzymanie nagłówków odpowiedzi z góry | -| `FETCH_BODY_TIMEOUT_MS` | dziedziczy `FETCH_TIMEOUT_MS` | Limit czasu Undici pomiędzy fragmentami treści powyżej (`0` wyłącza tę opcję) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Limit czasu połączenia TCP Undici | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici przekroczenie limitu czasu gniazda utrzymującego aktywność w stanie bezczynności | -| `TLS_CLIENT_TIMEOUT_MS` | dziedziczy `FETCH_TIMEOUT_MS` | Limit czasu dla żądań odcisków palców TLS złożonych przez `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | dziedziczy `REQUEST_TIMEOUT_MS` lub `30000` | Przekroczono limit czasu dla przekazywania proxy `/v1` z portu API do portu panelu kontrolnego | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Przekroczono limit czasu żądania przychodzącego na serwerze mostu API | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Przekroczono limit czasu nagłówka przychodzącego na serwerze mostu API | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Limit czasu utrzymywania aktywności na serwerze mostkowym API | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Przekroczono limit czasu braku aktywności gniazda na serwerze mostkowym API (`0` wyłącza tę funkcję) | +Advanced overrides are available if you need finer control: -Jeśli uruchamiasz OmniRoute za Nginx, Caddy, Cloudflare lub innym odwrotnym proxy, upewnij się, że proxy -limity czasu są również wyższe niż limity czasu transmisji/pobierania OmniRoute.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Otwórz Panel → `Dostawcy` i podłącz co najmniej jednego dostawcę (klucz OAuth lub API). -2. Otwórz Panel → `Punkty końcowe` i ​​utwórz klucz API. -3. (Opcjonalnie) Otwórz Panel → „Kombosy” i ustaw łańcuch awaryjny.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Współpracuje z Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode i pakietami SDK kompatybilnymi z OpenAI.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (dla operacji opartych na narzędziach):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -Następnie podłącz swojego klienta MCP przez `stdio` i przetestuj narzędzia takie jak: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` -- `kombinacje listy_omniroute_listy` +- `omniroute_list_combos` -**A2A (dla przepływów pracy między agentami):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -760,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Ten pakiet sprawdza rzeczywiste przepływy klientów MCP i A2A w porównaniu z działającą aplikacją.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -768,14 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` - +
+Void Linux (`xbps-src` template) -Unieważnij Linuksa (szablon `xbps-src`) - -Użytkownicy Void Linux mogą zbudować pakiet natywny za pomocą `xbps-src`. Zapisz ten blok jako `srcpkgs/omniroute/template`:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -787,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -795,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -871,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -882,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute jest dostępny jako publiczny obraz Dockera w [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Szybki bieg:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -892,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**Z plikiem środowiska:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Korzystanie z Docker Compose:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Obsługa pulpitu nawigacyjnego dla wdrożeń Dockera obejmuje teraz**Szybki tunel Cloudflare**dostępny jednym kliknięciem w `Pulpit nawigacyjny → Punkty końcowe`. Pierwsza opcja umożliwia pobieranie „cloudflared” tylko wtedy, gdy jest to konieczne, uruchamia tymczasowy tunel do bieżącego punktu końcowego „/v1” i wyświetla wygenerowany adres URL „https://\*.trycloudflare.com/v1” bezpośrednio pod normalnym publicznym adresem URL. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Uwagi: +Notes: -- Adresy URL szybkiego tunelu są tymczasowe i zmieniają się po każdym ponownym uruchomieniu. -- Szybkie tunele nie są automatycznie przywracane po ponownym uruchomieniu OmniRoute lub kontenera. W razie potrzeby włącz je ponownie z poziomu pulpitu nawigacyjnego. - — Instalacja zarządzana obsługuje obecnie systemy Linux, macOS i Windows na systemach `x64` / `arm64`. - — Zarządzane szybkie tunele domyślnie korzystają z transportu HTTP/2, aby uniknąć hałaśliwych ostrzeżeń o buforze QUIC UDP w ograniczonych środowiskach kontenerowych. Ustaw `CLOUDFLARED_PROTOCOL=quic` lub `auto`, jeśli chcesz inny transport. -- Obrazy platformy Docker łączą korzenie urzędu certyfikacji systemu i przekazują je do zarządzanej usługi „cloudflared”, co pozwala uniknąć błędów zaufania TLS podczas ładowania tunelu wewnątrz kontenera. -- SQLite działa w trybie WAL. Należy pozwolić na zakończenie `docker stop`, aby OmniRoute mógł sprawdzić najnowsze zmiany z powrotem do `storage.sqlite`. - — Dołączone pliki Compose mają już ustawiony okres karencji zatrzymania wynoszący 40 sekund. Jeśli uruchamiasz obraz bezpośrednio, zachowaj `--stop-timeout 40` (lub podobny), aby ręczne zatrzymanie nie przerwało czyszczenia przy zamykaniu. -- Ustaw `CLOUDFLARED_BIN=/absolute/path/to/cloudflared`, jeśli chcesz, aby OmniRoute użył istniejącego pliku binarnego zamiast go pobierać. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Korzystanie z Docker Compose w Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute można bezpiecznie ujawnić, korzystając z automatycznego udostępniania protokołu SSL Caddy. Upewnij się, że rekord DNS A Twojej domeny wskazuje adres IP Twojego serwera.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` +| Image | Tag | Size | Description | +| ------------------------ | -------- | ------ | --------------------- | +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | -| Obraz | Oznacz | Rozmiar | Opis | -| ------------------------ | -------- | ------ | ----------------------------------- | -| `diegosouzapw/omniroute` | `najnowsze` | ~250 MB | Najnowsza stabilna wersja | -| `diegosouzapw/omniroute` | `1.0.3` | ~250 MB | Aktualna wersja |--- +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**NOWOŚĆ!**OmniRoute jest teraz dostępny jako**natywna aplikacja komputerowa**dla systemów Windows, macOS i Linux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Uruchom OmniRoute jako samodzielną aplikację komputerową — bez terminala, bez przeglądarki i bez Internetu w przypadku modeli lokalnych. Aplikacja oparta na elektronach obejmuje: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Okno natywne**— dedykowane okno aplikacji z integracją z zasobnikiem systemowym -- 🔄**Auto-Start**— Uruchom OmniRoute po zalogowaniu się do systemu -- 🔔**Powiadomienia natywne**— otrzymuj powiadomienia o wyczerpaniu limitu lub problemach z dostawcami -- ⚡**Instalacja jednym kliknięciem**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Tryb offline**— Działa całkowicie offline z dołączonym serwerem### Szybki start +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Szybki start ```bash # Development mode @@ -981,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -Po zminimalizowaniu OmniRoute znajduje się w zasobniku systemowym i oferuje szybkie akcje: +When minimized, OmniRoute lives in your system tray with quick actions: -- Otwórz pulpit nawigacyjny -- Zmień port serwera -- Zamknij aplikację +- Open dashboard +- Change server port +- Quit application -📖 Pełna dokumentacja: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Poziom | Dostawca | Koszt | Reset przydziału | Najlepsze dla | -| ------------------ | ---------------------------- | ----------------------------------- | ----------------------------- | ---------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 SUBSKRYPCJA** | Claude Code (Pro) | 20 USD/mies. | 5h + tygodniowo | Już subskrybujesz | -| | Codex (Plus/Pro) | 20-200 $/mies. | 5h + tygodniowo | Użytkownicy OpenAI | -| | Bliźnięta CLI | **BEZPŁATNE** | 180 tys./mies. + 1 tys./dzień | Wszyscy! | -| | Drugi pilot GitHuba | 10–19 USD/mies. | Miesięczne | Użytkownicy GitHuba | -| **🔑 KLUCZ API** | NVIDIA NIM | **BEZPŁATNE**(na zawsze) | ~40 obr/min | Ponad 70 otwartych modeli | -| | Cerebra | **BEZPŁATNE**(1 mln tok/dzień) | 60 tys. TPM / 30 obr./min | Najszybszy na świecie | -| | Groq | **BEZPŁATNE**(30 obr./min) | 14,4 tys. RPD | Ultraszybka Lama/Gemma | -| | DeepSeek V3.2 | 0,27 USD/1,10 USD za 1 mln | Brak | Najlepsze uzasadnienie ceny/jakości | -| | xAI Grok-4 Szybki | **0,20 USD/0,50 USD za 1 milion**🆕 | Brak | Najszybsze + wywoływanie narzędzi, ultraniskie | -| | xAI Grok-4 (standard) | 0,20 USD/1,50 USD za 1 milion 🆕 | Brak | Rozsądny flagowiec od xAI | -| | Mistral | Bezpłatny okres próbny + płatny | Stawka ograniczona | Europejska sztuczna inteligencja | -| | OtwórzRouter | Płatność za użycie | Brak | Łącznie ponad 100 modeli | -| **💰 TANIO** | GLM-5 (przez Z.AI) 🆕 | 0,5 USD/1 mln | Codziennie 10:00 | Wyjście 128K, najnowszy flagowiec | -| | GLM-4.7 | 0,6 USD/1 mln | Codziennie 10:00 | Kopia zapasowa budżetu | -| | MiniMax M2.5 🆕 | Wejście 0,3 USD/1 mln | 5-godzinne toczenie | Rozumowanie + zadania agentyczne | -| | MiniMax M2.1 | 0,2 USD/1 mln | 5-godzinne toczenie | Najtańsza opcja | -| | Kimi K2.5 (Moonshot API) 🆕 | Płatność za użycie | Brak | Bezpośredni dostęp do API Moonshot | -| | Kimi K2 | 9 USD miesięcznie | 10 mln tokenów/mies. | Przewidywalny koszt | -| **🆓 DARMOWE** | Qoder | **$0** | Nieograniczony | 5 modeli bez ograniczeń | -| | Qwen | **$0** | Nieograniczony | 4 modele bez ograniczeń | -| | Kiro | **$0** | Nieograniczony | Claude Sonnet/Haiku (konstruktor AWS) | -| | LongCat Flash-Lite 🆕 | **0 $**(50 mln tok/dzień 🔥) | 1 RPS | Największy darmowy limit na Ziemi | -| | Pollinations AI 🆕 | **$0**(nie jest potrzebny klucz) | 1 zapotrzebowanie/15 s | GPT-5, Claude, DeepSeek, Lama 4 | -| | AI pracowników Cloudflare 🆕 | **0 $**(10 tys. neuronów/dzień) | ~150 odp/dzień | Ponad 50 modeli, globalna przewaga | -| | Scaleway AI 🆕 | **$0**(łącznie 1 milion tokenów) | Stawka ograniczona | UE/RODO, Qwen3 235B, Lama 70B | > 🆕**Dodano nowe modele (marzec 2026 r.):**Rodzina Grok-4 Fast w cenie 0,20 USD/0,50 USD/M (w porównaniu z czasem 1143 ms — 30% szybciej niż Gemini 2.5 Flash), GLM-5 przez Z.AI z mocą wyjściową 128 tys., rozumowanie MiniMax M2.5, zaktualizowana cena DeepSeek V3.2, Kimi K2.5 poprzez bezpośrednie API Moonshot. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 Stos kombinacji 0 $ — kompletna bezpłatna konfiguracja:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Zerowy koszt. Kodowanie nigdy się nie kończy.**Skonfiguruj to jako jedną kombinację OmniRoute, a wszystkie zmiany awaryjne będą odbywać się automatycznie — bez konieczności ręcznego przełączania.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Wszystkie poniższe modele są**100% bezpłatne i nie wymagają żadnej karty kredytowej**. OmniRoute automatycznie wyznacza trasy między nimi, gdy skończy się jeden limit — połącz je wszystkie, aby uzyskać niezniszczalną kombinację za 0 USD.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Modelka | Przedrostek | Limit | Limit stawki | -| ------------------- | ------ | --------- | ----------------------------------- | -| `claude-sonnet-4.5` | `kr/` |**Nieograniczony**| Brak raportu dziennego limitu | -| `claude-haiku-4.5` | `kr/` |**Nieograniczony**| Brak raportu dziennego limitu | -| `claude-opus-4.6` | `kr/` |**Nieograniczony**| Najnowsze Opus przez Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) -| Modelka | Przedrostek | Limit | Limit stawki | -| ------------------ | ------ | --------- | --------------- | -| `kimi-k2-myślenie` | `jeśli/` |**Nieograniczony**| Brak zgłoszonego limitu | -| `qwen3-koder-plus` | `jeśli/` |**Nieograniczony**| Brak zgłoszonego limitu | -| `głębokie wyszukiwanie-r1` | `jeśli/` |**Nieograniczony**| Brak zgłoszonego limitu | -| `minimax-m2.1` | `jeśli/` |**Nieograniczony**| Brak zgłoszonego limitu | -| `kimi-k2` | `jeśli/` |**Nieograniczony**| Brak zgłoszonego limitu | +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | --------------------- | +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -> Zalecana metoda połączenia:**Osobisty token dostępu + `qodercli`**. OAuth przeglądarki jest -> eksperymentalne i domyślnie wyłączone, chyba że skonfigurowano zmienne środowiskowe `QODER_OAUTH_*`.### 🟡 QWEN MODELS (Device Code Auth) +### 🟢 QODER MODELS (Free PAT via qodercli) -| Modelka | Przedrostek | Limit | Limit stawki | -| ------------------- | ------ | --------- | ------------------- | -| `qwen3-koder-plus` | `qw/` |**Nieograniczony**| Brak zgłoszonego limitu | -| `qwen3-koder-flash` | `qw/` |**Nieograniczony**| Brak zgłoszonego limitu | -| `qwen3-coder-next` | `qw/` |**Nieograniczony**| Brak zgłoszonego limitu | -| `model-wizji` | `qw/` |**Nieograniczony**| Multimodalny (zdjęcia) |### 🟣 GEMINI CLI (Google OAuth) +| Model | Prefix | Limit | Rate Limit | +| ------------------ | ------ | ------------- | --------------- | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -| Modelka | Przedrostek | Limit | Limit stawki | -| ------------------------ | ------ | ------------------------------------- | --------- | -| `podgląd-gemini-3-flash` | `gc/` |**180 tys. tok/miesiąc**+ 1 tys./dzień | Reset miesięczny | -| `gemini-2.5-pro` | `gc/` | 180 tys./miesiąc (wspólna pula) | Wysoka jakość |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| Poziom | Limit dzienny | Limit stawki | Notatki | +### 🟡 QWEN MODELS (Device Code Auth) + +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | ------------------- | +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | + +### 🟣 GEMINI CLI (Google OAuth) + +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | + +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---------- | ------------ | ----------- | ------------------------------------------------------ | -| Bezpłatny (wersja deweloperska) | Brak limitu tokenów |**~40 obr/min**| Ponad 70 modeli; przejście na czyste limity stawek w połowie 2025 r. | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -Popularne darmowe modele: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -| Poziom | Limit dzienny | Limit stawki | Notes | -| ---- | ------------------ | ---------------- | ------------------------------------------- | -| Bezpłatne |**1 mln tokenów dziennie**| 60 tys. TPM / 30 obr./min | Najszybsze na świecie wnioskowanie LLM; resetuje się codziennie | +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -Dostępne bezpłatnie: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-destill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -| Poziom | Limit dzienny | Limit stawki | Notatki | -| ---- | --------- | ---------------- | ----------------------------------------- | -| Bezpłatne |**14,4 tys. RPD**| 30 obr./min na model | Brak karty kredytowej; 429 w limicie, bez opłat | +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -Dostępne bezpłatnie: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +### 🔴 GROQ (Free API Key — console.groq.com) -| Modelka | Przedrostek | Dzienny bezpłatny limit | Notatki | -| ------------------------------ | ------ | ------------------ | ------------------------ | -| `LongCat-Flash-Lite` | `lc/` |**50 milionów tokenów**💥 | Największy darmowy limit w historii | -| `Czat Flash LongCat` | `lc/` | 500 tys. tokenów | Czat wieloobrotowy | -| „LongCat-Myślenie Flash” | `lc/` | 500 tys. tokenów | Rozumowanie / CoT | -| `LongCat-Flash-Myślenie-2601` | `lc/` | 500 tys. tokenów | Wersja ze stycznia 2026 r. | -| `LongCat-Flash-Omni-2603` | `lc/` | 500 tys. tokenów | Multimodalny | +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ------------- | ---------------- | ----------------------------------------- | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -> 100% za darmo w publicznej wersji beta. Zarejestruj się na [longcat.chat](https://longcat.chat) za pomocą e-maila lub telefonu. Resetuje się codziennie o godzinie 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -| Modelka | Przedrostek | Limit stawki | Dostawca za | +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 + +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | + +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. + +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| `openai` | `pol/` | 1 zapotrzebowanie/15 s | GPT-5 | -| ,,klaudiusz” | `pol/` | 1 zapotrzebowanie/15 s | Antropiczny Claude | -| `bliźniaki` | `pol/` | 1 zapotrzebowanie/15 s | Google Bliźnięta | -| „głębokie” | `pol/` | 1 zapotrzebowanie/15 s | DeepSeek V3 | -| „lama” | `pol/` | 1 zapotrzebowanie/15 s | Meta Lama 4 Zwiadowca | -| `mistral` | `pol/` | 1 zapotrzebowanie/15 s | Sztuczna inteligencja Mistrala | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**Zero tarcia:**Bez rejestracji, bez klucza API. Dodaj dostawcę Pollinations z pustym polem klucza i zacznie działać natychmiast.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| Poziom | Codzienne neurony | Równoważne użycie | Notatki | -| ---- | --------- | ---------------------------------------- | ------------------------ | -| Bezpłatne |**10 000**| ~150 LLM odpowiednio / 500 s audio / 15 000 osadzonych | Global edge, 50+ models | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 -Popularne darmowe modele: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (darmowe audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` +| Tier | Daily Neurons | Equivalent Usage | Notes | +| ---- | ------------- | --------------------------------------- | ----------------------- | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -> Wymaga tokena API + identyfikatora konta z [dash.cloudflare.com](https://dash.cloudflare.com). Zapisz identyfikator konta w ustawieniach dostawcy.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -| Poziom | Bezpłatny limit | Lokalizacja | Notatki | -| ---- | --------- | ------------ | ----------------------------------- | -| Bezpłatne |**1M tokenów**| 🇫🇷 Paryż, UE | W ramach limitów nie jest wymagana karta kredytowa | +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -Dostępne bezpłatnie: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 -> Zgodny z UE/RODO. Uzyskaj klucz API na [console.scaleway.com](https://console.scaleway.com). +| Tier | Free Quota | Location | Notes | +| ---- | ------------- | ------------ | ----------------------------------- | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | ->**💡 Najlepszy darmowy stos (11 dostawców, 0 $ na zawsze):** +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` + +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). + +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 mln tokenów dziennie 🔥 -> Zapylenia (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — klucz nie jest potrzebny -> Qwen (qw/) → modele qwen3-coder BEZ OGRANICZEŃ -> Gemini (gemini/) → Gemini 2.5 Flash — 1500 żądań dziennie za darmo -> Cloudflare AI (por./) → Ponad 50 modeli — 10 tys. neuronów dziennie -> Scaleway (scw/) → Qwen3 235B, Lama 70B — 1M darmowych tokenów (UE) -> Groq (groq/) → Lama/Gemma — ultraszybkie zapotrzebowanie 14,4 tys. dziennie -> NVIDIA NIM (nvidia/) → Ponad 70 otwartych modeli — 40 obr./min na zawsze -> Cerebras (cerebras/) → Lama/Qwen najszybsza na świecie — 1 mln tok/dzień -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Transkrypcja dowolnego audio/wideo za**0 USD**— Deepgram prowadzi z 200 USD za darmo, awaryjnie AssemblyAI 50 USD, Groq Whisper jako nieograniczona kopia zapasowa w sytuacjach awaryjnych. +## 🎙️ Free Transcription Combo -| Dostawca | Darmowe kredyty | Najlepszy Model | Limit stawki | -| ------------------ | -------------------------------- | -------------------------------------------- | ---------------------------- | -| 🟢**Deepgram**|**200 $ za darmo**(rejestracja) | `nova-3` — najlepsza dokładność, ponad 30 języków | Brak limitu obrotów na darmowe kredyty | -| 🔵**AssemblyAI**|**50 $ za darmo**(rejestracja) | `universal-3-pro` — rozdziały, sentyment, PII | Brak limitu obrotów na darmowe kredyty | -| 🔴**Grok**|**Za darmo na zawsze**| `whisper-large-v3` — szept OpenAI | 30 obr./min (ograniczona prędkość) | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. -**Sugerowana kombinacja w `/dashboard/combos`:**``` +| Provider | Free Credits | Best Model | Rate Limit | +| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | + +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Następnie w `/dashboard/media` → zakładka**Transkrypcja**: prześlij dowolny plik audio lub wideo → wybierz punkt końcowy combo → uzyskaj transkrypcję w obsługiwanych formatach.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 został zbudowany jako platforma operacyjna, a nie tylko pośrednik przekazujący.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Funkcja | Co to robi | -| --------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Grok-4 Szybka Rodzina** | Modele xAI w cenie 0,20 USD/0,50 USD/M — w teście porównawczym 1143 ms (30% szybsze niż Gemini 2.5 Flash) | -| 🧠**GLM-5 przez Z.AI** | Kontekst wyjściowy 128 tys., 0,5/1 mln USD — najnowszy flagowiec z rodziny GLM | -| 🔮**MiniMax M2.5** | Rozumowanie + zadania agenta za 0,30 USD/1 milion — znacząca aktualizacja z M2.1 | -| 🎯**ToolCalling Flag na model** | Dla każdego modelu „toolCalling: true/false” w rejestrze — AutoCombo pomija modele, które nie obsługują narzędzia | -| 🌍**Wielojęzyczne wykrywanie intencji** | Słowa kluczowe PT/ZH/ES/AR w punktacji AutoCombo — lepszy wybór modelu dla treści innych niż angielski | -| 📊**Rezerwacje oparte na testach** | Rzeczywiste opóźnienie p95 na podstawie punktacji kombinacji kanałów żądań na żywo — AutoCombo uczy się na podstawie rzeczywistych danych | -| 🔁**Poproś o deduplikację** | Okno deduplikacji oparte na skrótach treści — bezpieczne dla wielu agentów, zapobiega duplikacjom opłat | -| 🔌**Strategia routera z wtyczką** | Rozszerzalny interfejs `RouterStrategy` — dodaj niestandardową logikę routingu jako wtyczki | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Funkcja | Co to robi | -| -------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Modelowy plac zabaw** | Strona panelu kontrolnego do bezpośredniego testowania dowolnego modelu — selektory dostawcy/modelu/punktu końcowego, edytor Monako, przesyłanie strumieniowe, przerywanie, synchronizacja | -| 🔏**Dopasowanie odcisków palców CLI** | Kolejność nagłówków/treści poszczególnych dostawców zgodna z natywnymi podpisami CLI — przełącz według dostawcy w Ustawieniach > Zabezpieczenia.**Twój adres IP proxy zostanie zachowany** | -| 🤝**Wsparcie ACP (protokół klienta agenta)** | Wykrywanie agenta CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 innych), generator procesów, punkt końcowy `/api/acp/agents` | -| 🤖**Panel agentów AKP** | Debugowanie › Strona Agenty — siatka 14 agentów ze statusem instalacji, wersją i niestandardowym formularzem agenta dla dowolnego narzędzia CLI.**Użytkownicy OpenCode**otrzymują przycisk „Pobierz opencode.json”, który automatycznie generuje gotową do użycia konfigurację ze wszystkimi dostępnymi modelami. | -| 🔧**Routing w modelu niestandardowym `apiFormat`** | Niestandardowe modele z `apiFormat: „responses”` teraz poprawnie kierują do tłumacza API Responses | -| 🏢**Izolacja przestrzeni roboczej w Kodeksie** | Wiele obszarów roboczych Codex na e-mail — OAuth poprawnie rozdziela połączenia według identyfikatora obszaru roboczego | -| 🔄**Automatyczna aktualizacja elektronów** | Aplikacja komputerowa sprawdza dostępność aktualizacji + automatyczna instalacja przy ponownym uruchomieniu | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Funkcja | Co to robi | -| ----------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**Serwer MCP (25 narzędzi)** | Narzędzia IDE/agent za pośrednictwem 3 transportów: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 rdzeni + 3 pamięci + 4 narzędzia umiejętności | -| 🤝**Serwer A2A (JSON-RPC + SSE)** | Wykonywanie zadań między agentami z przepływami synchronizacji i przesyłania strumieniowego | -| 🧭**Strona skonsolidowanych punktów końcowych** | Strona zarządzania z kartami z zakładkami Endpoint Proxy, MCP, A2A i API Endpoints | -| 🎚️**Przełącza włączanie/wyłączanie usługi** | Przełączniki ON/OFF dla MCP i A2A z trwałością ustawień (domyślnie: OFF) | -| 🛰️**Bicie serca środowiska wykonawczego MCP** | Rzeczywisty status procesu (pid, czas pracy, wiek pulsu, transport, tryb zakresu) | -| 📋**Ścieżka audytu MCP** | Filtrowane dzienniki audytu z informacją o powodzeniu/porażce i przypisaniu klucza | -| 🔐**Egzekwowanie zakresu MCP** | 10 szczegółowych uprawnień zakresu dla kontrolowanego dostępu do narzędzi | -| 📡**Zarządzanie cyklem życia zadań A2A** | Wyświetlaj/filtruj zadania, sprawdzaj zdarzenia/artefakty, anuluj uruchomione zadania | -| 📋**Odkrycie karty agenta** | `/.well-known/agent.json` do automatycznego wykrywania klientów | -| 🧪**Uprząż testowa protokołu E2E** | Prawdziwy klient MCP SDK + A2A przepływa w `test:protocols:e2e` | -| ⚙️**Kontrola operacyjna** | Kombinacja przełączników, zastosuj profile odporności, zresetuj wyłączniki z jednej powierzchni sterującej | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Funkcja | Co to robi | -| ------------------------------------------------- | -------------------------------------------------------------------------------------- | ----------------------- | -| 🎯**Inteligentny 4-poziomowy powrót** | Auto-trasa: Subskrypcja → Klucz API → Tanie → Bezpłatne | -| 📊**Śledzenie limitów w czasie rzeczywistym** | Liczba tokenów na żywo + odliczanie resetowania dla każdego dostawcy | -| 🔄**Tłumaczenie formatu** | OpenAI ↔ Claude ↔ Gemini ↔ Odpowiedzi z konwersjami bezpiecznymi dla schematu | -| 👥**Obsługa wielu kont** | Wiele kont na dostawcę z inteligentnym wyborem | -| 🔄**Automatyczne odświeżanie tokena** | Tokeny OAuth odświeżają się automatycznie przy ponownej próbie | -| 🎨**Niestandardowe kombinacje** | 9 strategii równoważenia + kontrola łańcucha awaryjnego | -| 🌐**Router z dziką kartą** | `dostawca/*` routing dynamiczny | -| 🧠**Myślenie o kontroli budżetu** | Limity rozumowania passthrough, automatycznego, niestandardowego i adaptacyjnego | -| 🔀**Aliasy modeli** | Wbudowane + niestandardowe aliasing modelu i bezpieczeństwo migracji | -| ⚡**Degradacja tła** | Kieruj zadania w tle o niskim priorytecie do tańszych modeli | -| 🧪**Inteligentny routing uwzględniający zadania** | Automatyczny wybór modelu według rodzaju treści (kodowanie/wizja/analiza/podsumowanie) | -| 🔄**Przepływ pracy agenta A2A** | Deterministyczny orkiestrator FSM do stanowego, wieloetapowego wykonywania agentów | -| 🔀**Trasowanie adaptacyjne** | Zastąpienie strategii dynamicznej w oparciu o liczbę tokenów i złożoność komunikatu | -| 🎲**Różnorodność dostawców** | Punktacja entropii Shannona równoważąca automatyczną dystrybucję ruchu kombinowanego | -| 💬**Wstrzyknięcie monitu systemowego** | Globalna kontrola zachowania stosowana konsekwentnie | -| 📄**Zgodność API odpowiedzi** | Pełna obsługa `/v1/responses` dla Kodeksu i zaawansowanych przepływów pracy agentów | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Funkcja | Co to robi | -| -------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**Generowanie obrazu** | `/v1/images/generations` z backendem w chmurze i lokalnym | -| 📐**Osadzenia** | `/v1/embeddings` dla potoków wyszukiwania i RAG | -| 🎤**Transkrypcja audio** | `/v1/audio/transkrypcje` — 7 dostawców (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), automatyczne wykrywanie języka, obsługa MP4/MP3/WAV | -| 🔊**Zamiana tekstu na mowę** | `/v1/audio/speech` — 10 dostawców (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) z poprawnymi komunikatami o błędach | -| 🎬**Generowanie wideo** | `/v1/videos/generacje` (przepływy pracy ComfyUI + SD WebUI) | -| 🎵**Pokolenie muzyki** | `/v1/music/generacje` (przepływy pracy ComfyUI) | -| 🛡️**Moderacje** | Kontrole bezpieczeństwa `/v1/modations` | -| 🔀**Ponowna pozycja** | `/v1/rerank` dla punktacji trafności | -| 🔍**Wyszukiwarka internetowa**🆕 | `/v1/search` — 5 dostawców (Serper, Brave, Perplexity, Exa, Tavily), ponad 6500 darmowych miesięcznie, automatyczne przełączanie awaryjne, pamięć podręczna | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Funkcja | Co to robi | -| --------------------------------------------------- | --------------------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**Wyłączniki** | Wyłączenie/odzyskiwanie dla każdego modelu z kontrolą progów | -| 🎯**Modele obsługujące punkt końcowy** | Modele niestandardowe deklarują obsługiwane punkty końcowe + format API | -| 🛡️**Stado Przeciw Gromom** | Mutex + zabezpieczenia semaforów przy zdarzeniach ponawiania/oceniania | -| 🧠**Semantyczna + pamięć podręczna podpisów** | Redukcja kosztów/opóźnień dzięki dwóm warstwom pamięci podręcznej | -| ⚡**Poproś o idempotencję** | Zduplikowane okno ochrony | -| 🔒**Podrabianie odcisków palców TLS** | Odcisk palca TLS podobny do przeglądarki —**ogranicza wykrywanie botów i oznaczanie kont** | -| 🔏**Dopasowanie odcisków palców CLI** | Pasuje do natywnych sygnatur żądań CLI —**zmniejsza ryzyko bana, zachowując jednocześnie adres IP proxy** | -| 🌐**Filtrowanie IP** | Kontrola listy dozwolonych/blokowanych w przypadku ujawnionych wdrożeń | -| 📊**Edytowalne limity stawek** | Konfigurowalne limity na poziomie globalnym/dostawcy z trwałością | -| 📉**Wdzięku degradacja** | Rezerwowe możliwości wielowarstwowe chroniące podstawowe operacje bramy | -| 📜**Ścieżka audytu konfiguracji** | Śledzenie zmian w oparciu o różnice, zapobiegające dryfowaniu operacyjnemu dzięki prostym wycofywaniom | -| ⏳**Synchronizacja stanu dostawcy** | Proaktywne monitorowanie wygaśnięcia tokena wyzwalające alerty przed błędami autoryzacji | -| 🚪**Automatycznie wyłączaj zbanowane konta** | Wyłącznik operacyjny automatycznie zamyka trwale zablokowane konta tokenów | -| 🔑**Zarządzanie kluczami API + zakres** | Bezpieczne wydawanie/rotacja kluczy oraz kontrola modelu/dostawcy | -| 👁️**Ujawnienie klucza API o określonym zakresie**🆕 | Wyraź zgodę na odzyskiwanie kluczy API poprzez `ALLOW_API_KEY_REVEAL` | -| 🛡️**Chronione `/modelki`** | Opcjonalne bramkowanie uwierzytelniania i ukrywanie dostawców dla katalogu modeli | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Funkcja | Co to robi | -| ----------------------------------------------------- | --------------------------------------------------------------------------------------- | ---------------------------- | -| 📝**Żądanie + rejestrowanie proxy** | Pełne rejestrowanie żądań/odpowiedzi i proxy | -| 📉**Szczegółowe dzienniki przesyłane strumieniowo**🆕 | Rekonstruuje strumienie ładunku SSE w interfejsie użytkownika | -| 📋**Ujednolicony panel dzienników** | Widoki żądań, proxy, audytu i konsoli na jednej stronie | -| 🔍**Poproś o telemetrię** | Opóźnienie p50/p95/p99 i śledzenie żądań | -| 🏥**Panel zdrowia** | Czas pracy, stany wyłączników, blokady, statystyki pamięci podręcznej | -| 💰**Śledzenie kosztów** | Kontrola budżetu i widoczność cen dla poszczególnych modeli | -| 📈**Wizualizacje analityczne** | Informacje o użyciu modelu/dostawcy i widoki trendów | -| 🧪**Ramy oceny** | Testowanie złotego zestawu z konfigurowalnymi strategiami dopasowania | -| 📡**Diagnostyka na żywo**🆕 | Semantyczne obejście pamięci podręcznej w celu dokładnego testowania kombinacji na żywo | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Funkcja | Co to robi | -| ---------------------------------------------- | -------------------------------------------------------------------------------------- | --------------------- | -| 🌐**Wdrażaj gdziekolwiek** | Localhost, VPS, Docker, środowiska chmurowe | -| 🚇**Tunel Cloudflare**🆕 | Integracja Quick Tunnel jednym kliknięciem z poziomu pulpitu nawigacyjnego | -| 🔑**Filtrowanie modeli kluczy API** | Odpowiedź natywna /v1/models filtrowana według przypisanych ról kontekstowych nośnika | -| ⚡**Inteligentne obejście pamięci podręcznej** | Konfigurowalna heurystyka TTL i kontrola wymuszonego pobierania | -| 🔄**Kopia zapasowa/Przywracanie** | Przepływy eksportu/importu i odzyskiwania po awarii | -| 🧙**Kreator wdrażania** | Konfiguracja z przewodnikiem po pierwszym uruchomieniu | -| 🔧**Panel narzędzi CLI** | Konfiguracja popularnych narzędzi do kodowania jednym kliknięciem | -| 🎮**Modelowy plac zabaw** | Przetestuj dowolnego dostawcę/model/punkt końcowy z poziomu pulpitu nawigacyjnego | -| 🔏**Przełączanie linii papilarnych CLI** | Dopasowywanie odcisków palców poszczególnych dostawców w Ustawieniach > Bezpieczeństwo | -| 🌐**i18n (30 języków)** | Pełny pulpit nawigacyjny + obsługa języków dokumentów z pokryciem RTL | -| 🧹**Wyczyść wszystkie modele** | Czyszczenie listy modeli jednym kliknięciem w szczegółach dostawcy | -| 👁️**Sterowanie na pasku bocznym**🆕 | Ukryj komponenty i integracje w Ustawieniach wyglądu | -| 📋**Szablony wydania** | Standaryzowane szablony GitHub dla błędów i funkcji | -| 📂**Niestandardowy katalog danych** | Zastąpienie `DATA_DIR` miejsca przechowywania | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1294,105 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -W przypadku spadku limitu, szybkości lub kondycji OmniRoute automatycznie przechodzi do następnego kandydata bez konieczności ręcznego przełączania.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A można znaleźć w interfejsie użytkownika i dokumentach (nie są ukryte) -- Interfejsy API stanu protokołu udostępniają aktualne dane operacyjne (`/api/mcp/*`, `/api/a2a/*`) -- Pulpity nawigacyjne zawierają działania dla operacji dnia 2 (przełączniki kombinacji, resetowanie wyłączników, anulowanie zadań)#### Translator + validation workflow +#### Protocol management that is visible and operable -Obszar Tłumacza obejmuje: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Plac zabaw**: poproś o sprawdzenie transformacji -**Tester czatu**: pełne żądanie/odpowiedź w obie strony -**Stanowisko testowe**: wiele przypadków w jednym przebiegu -**Monitor na żywo**: podgląd ruchu drogowego w czasie rzeczywistym +#### Translator + validation workflow -Plus weryfikacja protokołu z prawdziwymi klientami poprzez `npm run test:protocols:e2e`. +The Translator area includes: -> 📖**[README serwera MCP](open-sse/mcp-server/README.md)**— Informacje o narzędziach, konfiguracje IDE i przykłady klientów +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[README serwera A2A](src/lib/a2a/README.md)**— Umiejętności, metody JSON-RPC, przesyłanie strumieniowe i cykl życia zadań## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute zawiera wbudowaną platformę ewaluacyjną do testowania jakości odpowiedzi LLM na podstawie złotego zestawu. Uzyskaj do niego dostęp poprzez**Analytics → Evals**na pulpicie nawigacyjnym.### Built-in Golden Set +## 🧪 Evaluations (Evals) -Fabrycznie załadowany „Złoty zestaw OmniRoute” zawiera przypadki testowe dla: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Pozdrowienia, matematyka, geografia, generowanie kodu -- Zgodność z formatem JSON, tłumaczenie, generowanie przecen -- Odmowa bezpieczeństwa (szkodliwa treść), liczenie, logika boolowska### Evaluation Strategies +### Built-in Golden Set -| Strategia | Opis | Przykład | -| ------------------- | ----------------------------------------------------------------------- | ----------------------------------- | --- | -| „dokładny” | Dane wyjściowe muszą dokładnie odpowiadać | ``4'` | -| `zawiera` | Dane wyjściowe muszą zawierać podciąg (wielkość liter nie ma znaczenia) | `„Paryż”` | -| wyrażenie regularne | Dane wyjściowe muszą pasować do wzorca wyrażenia regularnego | `"1.*2.*3"` | -| „niestandardowe” | Niestandardowa funkcja JS zwraca wartość prawda/fałsz | `(wyjście) => długość.wyjścia > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) - +
+🧩 MCP Setup (Model Context Protocol) -🧩 Konfiguracja MCP (protokół kontekstu modelu) +Start MCP transport in stdio mode: -Uruchom transport MCP w trybie stdio:```bash +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Zalecany przebieg walidacji: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Podłącz klienta MCP przez stdio. -2. Uruchom `omniroute_get_health`. -3. Uruchom `omniroute_list_combos`. -4. Otwórz `/dashboard/mcp`, aby potwierdzić puls, aktywność i audyt. +Useful APIs for automation: -Przydatne API do automatyzacji: +- `GET /api/mcp/status` +- `GET /api/mcp/tools` +- `GET /api/mcp/audit` +- `GET /api/mcp/audit/stats` -- `POBIERZ /api/mcp/status` -- `POBIERZ /api/mcp/tools` -- `POBIERZ /api/mcp/audit` -- `POBIERZ /api/mcp/audit/stats`
+ - -🤝 Konfiguracja A2A (Agent2Agent) +
+🤝 A2A Setup (Agent2Agent) -Odkryj agenta:```bash +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Wyślij zadanie:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` +Manage lifecycle: -Zarządzaj cyklem życia: - -- `POBIERZ /api/a2a/status` -- `POBIERZ /api/a2a/zadania` -- `POBIERZ /api/a2a/tasks/:id` +- `GET /api/a2a/status` +- `GET /api/a2a/tasks` +- `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -Interfejs operacyjny: +Operational UI: -- `/dashboard/a2a` dla obserwowalności zadania/stanu/strumienia i działań dymnych
+- `/dashboard/a2a` for task/state/stream observability and smoke actions - -🧪 Kompleksowa weryfikacja protokołu + -Sprawdź oba protokoły u prawdziwych klientów:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -To weryfikuje: +This verifies: -- Klient MCP SDK łączy/listuje/wywołuje -- Wykrywanie/wysyłanie/przesyłanie strumieniowe/pobieranie/anulowanie A2A -- Sprawdzaj dane w interfejsach API audytu MCP i zarządzania zadaniami A2A
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs - + -💳 Dostawcy subskrypcji### Claude Code (Pro/Max) +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1405,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Wskazówka dla profesjonalistów:**używaj Opus do skomplikowanych zadań, a Sonnet do szybkości. OmniRoute śledzi limit na model!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1419,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Każde konto Codex ma teraz przełączniki zasad w `Panel kontrolny -> Dostawcy`: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (ON/OFF): wymusza politykę 5-godzinnego progu okna. -- „Co tydzień” (WŁ./WYŁ.): wymusza politykę tygodniowego progu okna. -- Zachowanie progowe: gdy włączone okno osiągnie >=90% wykorzystania, to konto zostanie pominięte. -- Zachowanie rotacyjne: OmniRoute automatycznie kieruje do następnego kwalifikującego się konta Codex. -- Zachowanie resetowania: po upływie czasu „resetAt” dostawcy konto automatycznie staje się ponownie uprawnione. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Scenariusze: +Scenarios: -- `5h ON` + `Co tydzień ON`: konto jest pomijane, gdy którekolwiek okno osiągnie próg. -- `5h OFF` + `Co tydzień ON`: tylko cotygodniowe użycie może zablokować konto. -- `5h ON` + `Co tydzień OFF`: tylko 5-godzinne użytkowanie może zablokować konto. -- `resetAt` zaliczone: konto automatycznie ponownie wchodzi do rotacji (nie ma możliwości ręcznego ponownego włączenia).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1444,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Najlepsza wartość:**Ogromny darmowy poziom! Użyj tego przed płatnymi poziomami.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1459,74 +1662,91 @@ Models:
- +
+🔑 API Key Providers -🔑 Dostawcy kluczy API### NVIDIA NIM (FREE developer access — 70+ models) +### NVIDIA NIM (FREE developer access — 70+ models) -1. Zarejestruj się: [build.nvidia.com](https://build.nvidia.com) -2. Uzyskaj bezpłatny klucz API (w cenie 1000 kredytów) -3. Panel kontrolny → Dodaj dostawcę → NVIDIA NIM: - - Klucz API: `nvapi-your-key` +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Modele:**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` i ponad 50 innych +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -**Wskazówka dla profesjonalistów:**API zgodne z OpenAI — działa bezproblemowo z tłumaczeniem formatu OmniRoute!### DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -1. Zarejestruj się: [platform.deepseek.com](https://platform.deepseek.com) -2. Zdobądź klucz API -3. Panel kontrolny → Dodaj dostawcę → DeepSeek +### DeepSeek -**Modele:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -1. Zarejestruj się: [console.groq.com](https://console.groq.com) -2. Uzyskaj klucz API (w cenie bezpłatna warstwa) -3. Panel kontrolny → Dodaj dostawcę → Groq +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Modele:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +### Groq (Free Tier Available!) -**Wskazówka dla profesjonalistów:**Ultraszybkie wnioskowanie — najlepsze do kodowania w czasie rzeczywistym!### OpenRouter (100+ Models) +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -1. Zarejestruj się: [openrouter.ai](https://openrouter.ai) -2. Zdobądź klucz API -3. Panel kontrolny → Dodaj dostawcę → OpenRouter +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Modele:**Uzyskaj dostęp do ponad 100 modeli wszystkich głównych dostawców za pomocą jednego klucza API. +**Pro Tip:** Ultra-fast inference — best for real-time coding! -**Zachowanie panelu kontrolnego:**Modele OpenRouter są zarządzane z poziomu**Dostępnych modeli**. Ręczne dodawanie, importowanie i automatyczna synchronizacja aktualizują tę samą listę.
+### OpenRouter (100+ Models) - +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -💰 Tani dostawcy (kopia zapasowa)### GLM-4.7 (Daily reset, $0.6/1M) +**Models:** Access 100+ models from all major providers through a single API key. -1. Zarejestruj się: [Zhipu AI](https://open.bigmodel.cn/) -2. Uzyskaj klucz API z planu kodowania -3. Panel → Dodaj klucz API: - - Dostawca: `glm` - - Klucz API: `twój-klucz` +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -**Zastosuj:**`glm/glm-4.7` + -**Wskazówka dla profesjonalistów:**Plan kodowania oferuje 3× limit przy cenie 1/7! Resetuj codziennie o 10:00.### MiniMax M2.1 (5h reset, $0.20/1M) +
+💰 Cheap Providers (Backup) -1. Zarejestruj się: [MiniMax](https://www.minimax.io/) -2. Zdobądź klucz API -3. Panel → Dodaj klucz API +### GLM-4.7 (Daily reset, $0.6/1M) -**Zastosowanie:**`minimax/MiniMax-M2.1` +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**Wskazówka dla profesjonalistów:**Najtańsza opcja dla długiego kontekstu (1 milion tokenów)!### Kimi K2 ($9/month flat) +**Use:** `glm/glm-4.7` -1. Subskrybuj: [Moonshot AI](https://platform.moonshot.ai/) -2. Zdobądź klucz API -3. Panel → Dodaj klucz API +**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. -**Użyj:**`kimi/kimi-najnowsze` +### MiniMax M2.1 (5h reset, $0.20/1M) -**Wskazówka dla profesjonalistów:**Naprawiono 9 USD miesięcznie za 10 mln tokenów = efektywny koszt 0,90 USD/1 mln!
+1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key - +**Use:** `minimax/MiniMax-M2.1` -🆓 DARMOWE dostawcy (awaryjna kopia zapasowa)### Qoder (5 FREE models via OAuth) +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1567,9 +1787,10 @@ Models:
- +
+🎨 Create Combos -🎨 Twórz kombinacje### Example 1: Maximize Subscription → Cheap Backup +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1597,9 +1818,10 @@ Cost: $0 forever!
- +
+🔧 CLI Integration -🔧 Integracja CLI### Cursor IDE +### Cursor IDE ``` Settings → Models → Advanced: @@ -1610,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Użyj strony**Narzędzia CLI**w panelu kontrolnym, aby dokonać konfiguracji jednym kliknięciem, lub edytuj ręcznie plik `~/.claude/settings.json`.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1621,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Opcja 1 — Panel kontrolny (zalecany):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Opcja 2 — Ręcznie:**Edytuj `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1638,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Uwaga:**OpenClaw działa tylko z lokalnym OmniRoute. Użyj `127.0.0.1` zamiast `localhost`, aby uniknąć problemów z rozdzielczością IPv6.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1652,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Krok 1:**Dodaj OmniRoute jako niestandardowego dostawcę:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Krok 2:**Utwórz/edytuj plik `opencode.json` w katalogu głównym projektu:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1678,117 +1909,130 @@ opencode } } } -```` +``` -**Krok 3:**Wybierz model w OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Wskazówka:**Dodaj dowolny model dostępny w punkcie końcowym `/v1/models` OmniRoute do sekcji „modele”. Użyj formatu „dostawca/identyfikator modelu” w panelu kontrolnym OmniRoute.
+ --- ## Rozwiązywanie problemów - -Kliknij, aby rozwinąć przewodnik dotyczący rozwiązywania problemów +
+Click to expand troubleshooting guide -**„Model językowy nie dostarczał komunikatów”** +**"Language model did not provide messages"** -- Wyczerpany limit dostawcy → Sprawdź moduł śledzenia limitów na pulpicie nawigacyjnym -- Rozwiązanie: użyj kombinacji zastępczej lub przejdź na tańszy poziom +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**Ograniczenie szybkości** +**Rate limiting** -- Limit subskrypcji wyczerpany → Powrót do GLM/MiniMax -- Dodaj kombinację: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**Token OAuth wygasł** +**OAuth token expired** -- Automatyczne odświeżanie przez OmniRoute -- Jeśli problemy nadal występują: Panel kontrolny → Dostawca → Połącz ponownie +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Wysokie koszty** +**High costs** -- Sprawdź statystyki użytkowania w Panelu → Koszty -- Zmień model podstawowy na GLM/MiniMax -- Używaj bezpłatnej warstwy (Gemini CLI, Qoder) do zadań niekrytycznych +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Porty pulpitu nawigacyjnego/API są nieprawidłowe** +**Dashboard/API ports are wrong** -- `PORT` to kanoniczny port bazowy (domyślnie port API) -- `API_PORT` zastępuje tylko odbiornik API zgodny z OpenAI -- `DASHBOARD_PORT` zastępuje tylko pulpit nawigacyjny/nasłuchiwanie Next.js -- Ustaw „NEXT_PUBLIC_BASE_URL” na swój pulpit nawigacyjny/publiczny adres URL (w przypadku wywołań zwrotnych OAuth) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Błędy synchronizacji z chmurą** +**Cloud sync errors** -- Sprawdź, czy `BASE_URL` wskazuje na działającą instancję -- Sprawdź, czy `CLOUD_URL` wskazuje na oczekiwany punkt końcowy w chmurze -- Zachowaj wyrównanie wartości `NEXT_PUBLIC_*` z wartościami po stronie serwera +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Pierwsze logowanie nie działa** +**First login not working** -- Sprawdź `INITIAL_PASSWORD` w `.env` -- Jeśli nieustawione, hasło zastępcze to `123456` +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Brak dzienników żądań** +**No request logs** -- Artefakty żądań są zapisywane w `DATA_DIR/call_logs/` jako jeden plik JSON na żądanie -- Włącz przechwytywanie potoku z Panelu sterowania → Dzienniki → Dzienniki żądań, jeśli potrzebujesz szczegółowych ładunków na każdym etapie -- Ustaw `APP_LOG_TO_FILE=true`, jeśli chcesz także rejestrować logi konsoli aplikacji w `logs/application/app.log` -- Dostosuj `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` i `CALL_LOG_MAX_ENTRIES` w razie potrzeby +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Test połączenia pokazuje „Nieprawidłowy” dla dostawców kompatybilnych z OpenAI** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Wielu dostawców nie ujawnia punktu końcowego `/models` -- OmniRoute v1.0.6+ zawiera weryfikację awaryjną poprzez uzupełnianie czatu -- Upewnij się, że podstawowy adres URL zawiera przyrostek `/v1`### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ Ważne dla użytkowników korzystających z OmniRoute na VPS, Dockerze lub dowolnym serwerze zdalnym**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -Dostawcy**Antigravity**i**Gemini CLI**korzystają z**Google OAuth 2.0**. Google wymaga, aby parametr „redirect_uri” w przepływie OAuth dokładnie odpowiadał jednemu z wcześniej zarejestrowanych identyfikatorów URI w Google Cloud Console aplikacji. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -Poświadczenia OAuth zawarte w OmniRoute są rejestrowane**tylko dla `localhost`**. Gdy uzyskujesz dostęp do OmniRoute na zdalnym serwerze (np. `https://omniroute.myserver.com`), Google odrzuca uwierzytelnienie za pomocą:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Musisz utworzyć**Identyfikator klienta OAuth 2.0**w Google Cloud Console przy użyciu identyfikatora URI swojego serwera.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Otwórz konsolę Google Cloud** +#### Step-by-step -Przejdź do: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Utwórz nowy identyfikator klienta OAuth 2.0** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Kliknij**"+ Utwórz dane uwierzytelniające"**→**"Identyfikator klienta OAuth"** -- Typ aplikacji:**"Aplikacja internetowa"** -- Nazwa: dowolna (np. „OmniRoute Remote”) +**2. Create a new OAuth 2.0 Client ID** -**3. Dodaj autoryzowane identyfikatory URI przekierowań** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -W polu**„Autoryzowane identyfikatory URI przekierowań”**dodaj:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Zamień `twój-serwer.com` na domenę lub adres IP swojego serwera (w razie potrzeby podaj port, np. `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. Zapisz i skopiuj dane uwierzytelniające** +After creating, Google will show the **Client ID** and **Client Secret**. -Po utworzeniu Google wyświetli**Identyfikator klienta**i**Tajemnica klienta**. +**5. Set environment variables** -**5. Ustaw zmienne środowiskowe** +In your `.env` (or Docker environment variables): -W pliku `.env` (lub zmiennych środowiskowych Dockera):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1797,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Uruchom ponownie OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Spróbuj połączyć się ponownie** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Panel kontrolny → Dostawcy → Antygrawitacja (lub Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google będzie teraz poprawnie przekierowywać do `https://your-server.com/callback`.--- +--- #### Temporary workaround (without custom credentials) -Jeśli nie chcesz teraz konfigurować własnych danych uwierzytelniających, nadal możesz skorzystać z**ręcznego przepływu adresów URL**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute otwiera adres URL autoryzacji Google -2. Po autoryzacji Google próbuje przekierować do `localhost` (co kończy się niepowodzeniem na serwerze zdalnym) -3.**Skopiuj pełny adres URL**z paska adresu przeglądarki (nawet jeśli strona się nie ładuje) -4. Wklej ten adres URL w polu pokazanym w trybie połączenia OmniRoute -5. Kliknij**„Połącz”** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Działa to, ponieważ kod autoryzacyjny w adresie URL jest ważny niezależnie od tego, czy załadowana została strona przekierowująca.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. - -🇧🇷 Versão em Português#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Sprawdzone**Antigravity**i**Gemini CLI**używane**Google OAuth 2.0**dla autentyczności. O Google, możesz użyć `redirect_uri` bez zmiany OAuth seja**exatamente**uma das URI pre-cadastradas bez Google Cloud Console do aplikacji. +
+🇧🇷 Versão em Português -Jako uwierzytelnienie OAuth przypisane do OmniRoute jest cadastradas**apenas para `localhost`**. Możesz uzyskać dostęp do OmniRoute na serwerze zdalnym (np.: `https://omniroute.meuservidor.com`), lub Google rejeita a autenticação com:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Precyzyjne żądanie**Identyfikator klienta OAuth 2.0**nie Google Cloud Console przez serwer URI.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Dostęp do konsoli Google Cloud** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -**2. Wezwij nowy identyfikator klienta OAuth 2.0** +**2. Crie um novo OAuth 2.0 Client ID** -- Kliknij**"+ Utwórz dane uwierzytelniające"**→**"Identyfikator klienta OAuth"** -- Typ aplikacji:**"Aplikacja internetowa"** -- Nazwa: escolha qualquer nome (np. `OmniRoute Remote`) +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Adicione jako autoryzowane identyfikatory URI przekierowań** +**3. Adicione as Authorized Redirect URIs** -Bez komentarza**„Autoryzowane identyfikatory URI przekierowań”**, rada:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Substitua `seu-servidor.com` pelo domínio lub IP do seu servidor (w tym porta se necessário, np.: `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. Zapisz i skopiuj jako poświadczenie** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Após criar, o Google mostrará o**Identyfikator klienta**i o**Tajemnica klienta**. +**5. Configure as variáveis de ambiente** -**5. Skonfiguruj jako variáveis de ambiente** +No seu `.env` (ou nas variáveis de ambiente do Docker): -Nie seu `.env` (ou nas variáveis de ambiente do Docker):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1876,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reinicie lub OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute - -```` +``` **7. Tente conectar novamente** -Panel kontrolny → Dostawcy → Antygrawitacja (lub Gemini CLI) → OAuth +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Agora o Google redirecionará corretamente dla `https://seu-servidor.com/callback` i autenticação funcionará.--- +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. + +--- #### Workaround temporário (sem configurar credenciais próprias) -Jeśli chcesz uzyskać dostęp do**podręcznika URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. OmniRoute abrirá adres URL autoryzacji w Google +1. O OmniRoute abrirá a URL de autorização do Google 2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) -3.**Skopiuj kompletny adres URL**da barra de endereço do seu przeglądarki (wiadomość que a página não carregue) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) 4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute -5. Kliknij je**„Połącz”** +5. Clique em **"Connect"** -> To obejście funkcji porque o kodigo de autorização na URL é válido niezależny do przekierowania ter carregado ou não.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1914,64 +2171,73 @@ Jeśli chcesz uzyskać dostęp do**podręcznika URL**: ## 🛠️ Tech Stack - -Kliknij, aby rozwinąć szczegóły stosu technologii +
+Click to expand tech stack details --**Środowisko wykonawcze**: Node.js 18–22 LTS (⚠️ Node.js 24+ jest**nieobsługiwany**— natywne pliki binarne `better-sqlite3` są niekompatybilne) --**Język**: TypeScript 5.9 —**100% TypeScript**w `src/` i `open-sse/` (zero `any` w modułach podstawowych od wersji 2.0) --**Framework**: Next.js 16 + React 19 + Tailwind CSS 4 --**Baza danych**: LowDB (JSON) + SQLite (stan domeny + logi proxy + audyt MCP + decyzje dotyczące routingu) --**Schematy**: Zod (walidacja we/wy narzędzia MCP, kontrakty API) --**Protokoły**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Streaming**: zdarzenia wysyłane przez serwer (SSE) --**Auth**: OAuth 2.0 (PKCE) + JWT + klucze API + autoryzacja w zakresie MCP --**Testowanie**: Uruchomienie testów Node.js + Vitest (ponad 900 testów, w tym testy jednostkowe, integracyjne, E2E) --**CI/CD**: Akcje GitHub (automatyczne publikowanie npm + Docker Hub w momencie wydania) --**Strona internetowa**: [omniroute.online](https://omniroute.online) --**Pakiet**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Odporność**: wyłącznik automatyczny, wykładnicze wycofywanie, stado przeciwgrzmotowe, fałszowanie TLS, samonaprawianie automatycznej kombinacji
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Dokumentacja -| Dokument | Opis | +| Document | Description | | ---------------------------------------------- | --------------------------------------------------- | -| [Podręcznik użytkownika](docs/USER_GUIDE.md) | Dostawcy, kombinacje, integracja CLI, wdrożenie | -| [Dokumentacja API](docs/API_REFERENCE.md) | Wszystkie punkty końcowe z przykładami | -| [Serwer MCP](open-sse/mcp-server/README.md) | 16 narzędzi MCP, konfiguracje IDE, klienci Python/TS/Go | -| [Serwer A2A](src/lib/a2a/README.md) | Protokół JSON-RPC 2.0, umiejętności, streaming, zarządzanie zadaniami | -| [Silnik Auto-Combo](docs/auto-combo.md) | Punktacja 6-czynnikowa, pakiety trybów, samoleczenie | -| [Rozwiązywanie problemów](docs/TROUBLESHOOTING.md) | Typowe problemy i rozwiązania | -| [Architektura](docs/ARCHITECTURE.md) | Architektura systemu i elementy wewnętrzne | -| [Wspieranie](CONTRIBUTING.md) | Konfiguracja i wytyczne dotyczące programowania | -| [Specyfikacja OpenAPI](docs/openapi.yaml) | Specyfikacja OpenAPI 3.0 | -| [Polityka bezpieczeństwa](SECURITY.md) | Zgłaszanie luk w zabezpieczeniach i praktyki bezpieczeństwa | -| [Wdrożenie maszyny wirtualnej](docs/VM_DEPLOYMENT_GUIDE.md) | Kompletny przewodnik: konfiguracja VM + nginx + Cloudflare | -| [Galeria funkcji](docs/FEATURES.md) | Wizualna wycieczka po panelu ze zrzutami ekranu | -| [Lista kontrolna wydania](docs/RELEASE_CHECKLIST.md) | Etapy sprawdzania poprawności przed wydaniem |--- +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute ma**ponad 210 funkcji zaplanowanych**w wielu fazach rozwoju. Oto kluczowe obszary: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Kategoria | Planowane funkcje | Najważniejsze | -| ------------------------------ | ---------------- | ------------------------------------------------------------------------------------------------ | -| 🧠**Routing i inteligencja**| 25+ | Routing z najmniejszym opóźnieniem, routing oparty na tagach, wstępna inspekcja przydziału, wybór konta P2C | -| 🔒**Bezpieczeństwo i zgodność**| 20+ | Wzmocnienie SSRF, maskowanie poświadczeń, limit szybkości na punkt końcowy, zakres kluczy zarządzania | -| 📊**Obserwowalność**| 15+ | Integracja OpenTelemetry, monitorowanie kwot w czasie rzeczywistym, śledzenie kosztów według modelu | -| 🔄**Integracja dostawców**| 20+ | Rejestr modeli dynamicznych, czasy odnowienia dostawcy, Kodeks dla wielu kont, analiza przydziału Copilot | -| ⚡**Wydajność**| 15+ | Podwójna warstwa pamięci podręcznej, pamięć podręczna podpowiedzi, pamięć podręczna odpowiedzi, utrzymywanie transmisji strumieniowej, wsadowe API | -| 🌐**Ekosystem**| 10+ | WebSocket API, ładowanie konfiguracji na gorąco, rozproszony magazyn konfiguracji, tryb komercyjny |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**Integracja OpenCode**— natywna obsługa dostawców dla IDE kodowania OpenCode AI -- 🔗**Integracja z TRAE**— Pełne wsparcie dla platformy rozwojowej TRAE AI -- 📦**Batch API**— Asynchroniczne przetwarzanie wsadowe dla żądań masowych -- 🎯**Routing oparty na tagach**— Kieruj żądania na podstawie niestandardowych tagów i metadanych -- 💰**Strategia najniższych kosztów**— Automatycznie wybierz najtańszego dostępnego dostawcę +### 🔜 Coming Soon -> 📝 Pełna specyfikacja funkcji dostępna w [`docs/new-features/`](docs/new-features/) (217 szczegółowych specyfikacji)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1979,18 +2245,20 @@ OmniRoute ma**ponad 210 funkcji zaplanowanych**w wielu fazach rozwoju. Oto klucz ### How to Contribute -1. Forkuj repozytorium -2. Utwórz gałąź funkcji (`git checkout -b feature/amazing-feature`) -3. Zatwierdź zmiany („git commit -m 'Dodaj niesamowitą funkcję'`) -4. Push do gałęzi („git push origin feature/amazing-feature”) -5. Otwórz żądanie ściągnięcia +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Szczegółowe wytyczne można znaleźć w [CONTRIBUTING.md](CONTRIBUTING.md).### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -2002,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Specjalne podziękowania dla**[9router](https://github.com/decolua/9router)**autorstwa**[decolua](https://github.com/decolua)**— oryginalnego projektu, który zainspirował ten widelec. OmniRoute opiera się na tym niesamowitym fundamencie dzięki dodatkowym funkcjom, wielomodalnym interfejsom API i pełnemu przepisaniu TypeScriptu. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Specjalne podziękowania dla**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— oryginalnej implementacji Go, która zainspirowała ten port JavaScript.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Licencja -Licencja MIT — szczegółowe informacje można znaleźć w [LICENCJA](LICENCJA).--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/pl/docs/ARCHITECTURE.md b/docs/i18n/pl/docs/ARCHITECTURE.md index 5627977b4d..42c05dae89 100644 --- a/docs/i18n/pl/docs/ARCHITECTURE.md +++ b/docs/i18n/pl/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Ostatnia aktualizacja: 28.03.2026_## Executive Summary -OmniRoute to lokalna brama routingu AI i pulpit nawigacyjny zbudowany w oparciu o Next.js. -Zapewnia pojedynczy punkt końcowy zgodny z OpenAI (`/v1/*`) i kieruje ruch do wielu dostawców nadrzędnych z tłumaczeniem, rezerwą, odświeżaniem tokenów i śledzeniem użycia. -Podstawowe możliwości: +_Last updated: 2026-03-28_ -- Powierzchnia API kompatybilna z OpenAI dla CLI/narzędzi (28 dostawców) -- Tłumaczenie żądań/odpowiedzi w różnych formatach dostawców -- Awaryjna kombinacja modeli (sekwencja wielu modeli) -- Rezerwa awaryjna na poziomie konta (wiele kont na dostawcę) -- Zarządzanie połączeniem dostawcy klucza OAuth + API -- Generowanie osadzania poprzez `/v1/embeddings` (6 dostawców, 9 modeli) -- Generowanie obrazu poprzez `/v1/images/generations` (4 dostawców, 9 modeli) -- Pomyśl o analizie tagów (`...`) dla modeli rozumowania -- Oczyszczanie odpowiedzi w celu zapewnienia ścisłej zgodności z OpenAI SDK -- Normalizacja ról (programista → system, system → użytkownik) w celu zapewnienia zgodności między dostawcami -- Strukturalna konwersja danych wyjściowych (json_schema → Gemini respondSchema) -- Lokalna trwałość dostawców, kluczy, aliasów, kombinacji, ustawień, cen -- Śledzenie wykorzystania/kosztów i rejestrowanie żądań -- Opcjonalna synchronizacja w chmurze dla synchronizacji wielu urządzeń/stanów -- Lista dozwolonych/blokowanych adresów IP do kontroli dostępu do API -- Myślenie o zarządzaniu budżetem (przejściowe/automatyczne/niestandardowe/adaptacyjne) -- Globalny system natychmiastowego wstrzyknięcia -- Śledzenie sesji i pobieranie odcisków palców -- Ulepszone ograniczenie stawek dla konta z profilami specyficznymi dla dostawcy -- Wzór wyłącznika zapewniający odporność dostawcy -- Ochrona stada przed piorunami z blokadą mutex -- Pamięć podręczna deduplikacji żądań oparta na sygnaturach -- Warstwa domeny: dostępność modelu, zasady kosztów, polityka awaryjna, polityka blokad -- Trwałość stanu domeny (pamięć podręczna zapisu SQLite dla błędów awaryjnych, budżetów, blokad, wyłączników automatycznych) -- Silnik polityki do scentralizowanej oceny wniosków (blokada → budżet → rezerwa) - — Żądaj telemetrii z agregacją opóźnień p50/p95/p99 -- Identyfikator korelacji (X-Request-Id) do śledzenia od końca do końca -- Rejestrowanie audytu zgodności z możliwością rezygnacji dla każdego klucza API -- Ramy ewaluacyjne dla zapewnienia jakości LLM -- Pulpit nawigacyjny interfejsu użytkownika Resilience ze statusem wyłącznika automatycznego w czasie rzeczywistym -- Modułowi dostawcy OAuth (12 indywidualnych modułów w `src/lib/oauth/providers/`) +## Executive Summary -Podstawowy model środowiska wykonawczego: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- Trasy aplikacji Next.js w `src/app/api/*` implementują zarówno interfejsy API pulpitu nawigacyjnego, jak i interfejsy API zgodności -- Współdzielony rdzeń SSE/routingowy w `src/sse/*` + `open-sse/*` obsługuje wykonywanie dostawcy, tłumaczenie, przesyłanie strumieniowe, rezerwę i użycie## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Środowisko wykonawcze bramy lokalnej -- Interfejsy API zarządzania pulpitem nawigacyjnym -- Uwierzytelnianie dostawcy i odświeżanie tokena -- Poproś o tłumaczenie i przesyłanie strumieniowe SSE -- Stan lokalny + trwałość użytkowania -- Opcjonalna orkiestracja synchronizacji w chmurze### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Wdrożenie usługi w chmurze za `NEXT_PUBLIC_CLOUD_URL` -- Umowa SLA dostawcy/płaszczyzna kontroli poza procesem lokalnym -- Same zewnętrzne pliki binarne CLI (Claude CLI, Codex CLI itp.)## Dashboard Surface (Current) +### Out of Scope -Strony główne w `src/app/(dashboard)/dashboard/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — szybki start + przegląd dostawców -- `/dashboard/endpoint` — proxy punktu końcowego + MCP + A2A + zakładki punktu końcowego API -- `/dashboard/providers` — połączenia z dostawcami i dane uwierzytelniające -- `/dashboard/combos` — strategie combo, szablony, reguły routingu modelu -- `/dashboard/costs` — agregacja kosztów i widoczność cen -- `/dashboard/analytics` — analityka i ocena użytkowania -- `/dashboard/limits` — kontrola kwot/stawek -- `/dashboard/cli-tools` — wdrażanie CLI, wykrywanie środowiska wykonawczego, generowanie konfiguracji -- `/dashboard/agents` — wykryto agentów ACP + niestandardową rejestrację agenta -- `/dashboard/media` — plac zabaw dla obrazów/wideo/muzyki -- `/dashboard/search-tools` — testowanie i historia dostawcy wyszukiwania -- `/dashboard/health` — czas pracy, wyłączniki automatyczne, limity szybkości -- `/dashboard/logs` — logi żądań/proxy/audytu/konsoli -- `/dashboard/settings` — zakładki ustawień systemowych (ogólne, routing, domyślne kombinacje itp.) -- `/dashboard/api-manager` — Cykl życia klucza API i uprawnienia modelu## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Główne katalogi: +Main directories: -- `src/app/api/v1/*` i `src/app/api/v1beta/*` dla interfejsów API zgodności -- `src/app/api/*` dla interfejsów API zarządzania/konfiguracji -- Następnie przepisuje mapę `/v1/*` w `next.config.mjs` na `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Ważne ścieżki kompatybilności: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` — zawiera niestandardowe modele z opcją `custom: true` -- `src/app/api/v1/embeddings/route.ts` — generacja osadzania (6 dostawców) -- `src/app/api/v1/images/generations/route.ts` — generowanie obrazów (4+ dostawców, w tym Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[dostawca]/chat/completions/route.ts` — dedykowany czat dla każdego dostawcy -- `src/app/api/v1/providers/[dostawca]/embeddings/route.ts` — dedykowane osadzanie dla każdego dostawcy -- `src/app/api/v1/providers/[dostawca]/images/generations/route.ts` — obrazy dedykowane dla poszczególnych dostawców +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` -- `src/app/api/v1beta/models/[...ścieżka]/trasa.ts` +- `src/app/api/v1beta/models/[...path]/route.ts` -Domeny zarządzania: +Management domains: -- Auth/ustawienia: `src/app/api/auth/*`, `src/app/api/settings/*` -- Dostawcy/połączenia: `src/app/api/providers*` -- Węzły dostawcy: `src/app/api/provider-nodes*` -- Modele niestandardowe: `src/app/api/provider-models` (GET/POST/DELETE) -- Katalog modeli: `src/app/api/models/route.ts` (GET) -- Konfiguracja proxy: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- Klucze/aliasy/combo/ceny: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- Użycie: `src/app/api/usage/*` -- Synchronizacja/chmura: `src/app/api/sync/*`, `src/app/api/cloud/*` -- Pomocnicy narzędzi CLI: `src/app/api/cli-tools/*` -- Filtr IP: `src/app/api/settings/ip-filter` (GET/PUT) -- Myślący budżet: `src/app/api/settings/thinking-budget` (GET/PUT) -- Monit systemowy: `src/app/api/settings/system-prompt` (GET/PUT) -- Sesje: `src/app/api/sessions` (GET) -- Limity szybkości: `src/app/api/rate-limits` (GET) -- Resilience: `src/app/api/resilience` (GET/PATCH) — profile dostawców, wyłącznik, stan limitu szybkości -- Reset odporności: `src/app/api/resilience/reset` (POST) — resetowanie wyłączników + czasów odnowienia -- Statystyki pamięci podręcznej: `src/app/api/cache/stats` (GET/DELETE) -- Dostępność modelu: `src/app/api/models/availability` (GET/POST) -- Telemetria: `src/app/api/telemetry/summary` (GET) -- Budżet: `src/app/api/usage/budget` (GET/POST) -- Łańcuchy awaryjne: `src/app/api/fallback/chains` (GET/POST/DELETE) -- Audyt zgodności: `src/app/api/compliance/audit-log` (GET) +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) - Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- Zasady: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Policies: `src/app/api/policies` (GET/POST) -Główne moduły przepływowe: +## 2) SSE + Translation Core -- Wpis: `src/sse/handlers/chat.ts` -- Podstawowa orkiestracja: `open-sse/handlers/chatCore.ts` -- Adaptery wykonawcze dostawcy: `open-sse/executors/*` -- Wykrywanie formatu/konfiguracja dostawcy: `open-sse/services/provider.ts` -- Analiza/rozwiązanie modelu: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Logika rezerwowa konta: `open-sse/services/accountFallback.ts` -- Rejestr tłumaczeń: `open-sse/translator/index.ts` -- Transformacje strumieni: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- Ekstrakcja/normalizacja użycia: `open-sse/utils/usageTracking.ts` -- Pomyśl o parserze tagów: `open-sse/utils/thinkTagParser.ts` -- Procedura osadzania: `open-sse/handlers/embeddings.ts` -- Rejestr dostawców osadzania: `open-sse/config/embeddingRegistry.ts` -- Procedura obsługi generowania obrazu: `open-sse/handlers/imageGeneration.ts` -- Rejestr dostawców obrazów: `open-sse/config/imageRegistry.ts` -- Odkażanie odpowiedzi: `open-sse/handlers/responseSanitizer.ts` -- Normalizacja ról: `open-sse/services/roleNormalizer.ts` +Main flow modules: -Usługi (logika biznesowa): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- Wybór konta/punktacja: `open-sse/services/accountSelector.ts` -- Zarządzanie cyklem życia kontekstu: `open-sse/services/contextManager.ts` -- Wymuszanie filtra IP: `open-sse/services/ipFilter.ts` -- Śledzenie sesji: `open-sse/services/sessionManager.ts` -- Zażądaj deduplikacji: `open-sse/services/signatureCache.ts` -- Wstrzyknięcie monitu systemowego: `open-sse/services/systemPrompt.ts` -- Myślenie o zarządzaniu budżetem: `open-sse/services/thinkingBudget.ts` -- Routing modelu Wildcard: `open-sse/services/wildcardRouter.ts` -- Zarządzanie limitami stawek: `open-sse/services/rateLimitManager.ts` -- Wyłącznik automatyczny: `open-sse/services/circuitBreaker.ts` +Services (business logic): -Moduły warstwy domeny: +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- Dostępność modelu: `src/lib/domain/modelAvailability.ts` -- Reguły kosztów/budżety: `src/lib/domain/costRules.ts` -- Polityka awaryjna: `src/lib/domain/fallbackPolicy.ts` -- Funkcja rozpoznawania kombinacji: `src/lib/domain/comboResolver.ts` -- Polityka blokowania: `src/lib/domain/lockoutPolicy.ts` -- Silnik polityki: `src/domain/policyEngine.ts` — scentralizowana blokada → budżet → ocena rezerwowa -- Katalog kodów błędów: `src/lib/domain/errorCodes.ts` -- Identyfikator żądania: `src/lib/domain/requestId.ts` -- Limit czasu pobierania: `src/lib/domain/fetchTimeout.ts` -- Żądanie telemetrii: `src/lib/domain/requestTelemetry.ts` -- Zgodność/audyt: `src/lib/domain/compliance/index.ts` -- Biegacz Eval: `src/lib/domain/evalRunner.ts` -- Trwałość stanu domeny: `src/lib/db/domainState.ts` — SQLite CRUD dla łańcuchów awaryjnych, budżetów, historii kosztów, stanu blokady, wyłączników automatycznych +Domain layer modules: -Moduły dostawcy OAuth (12 pojedynczych plików w `src/lib/oauth/providers/`): +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -- Indeks rejestru: `src/lib/oauth/providers/index.ts` -- Dostawcy indywidualni: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` -- Cienki wrapper: `src/lib/oauth/providers.ts` — reeksport z poszczególnych modułów## 3) Persistence Layer +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -Baza danych stanu podstawowego (SQLite): +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -- Core infra: `src/lib/db/core.ts` (lepsze-sqlite3, migracje, WAL) -- Reeksport fasady: `src/lib/localDb.ts` (cienka warstwa kompatybilności dla osób wywołujących) -- plik: `${DATA_DIR}/storage.sqlite` (lub `$XDG_CONFIG_HOME/omniroute/storage.sqlite`, gdy jest ustawiony, w przeciwnym razie `~/.omniroute/storage.sqlite`) -- encje (tabele + przestrzenie nazw KV): dostawcaConnections, ProvideNodes, modelAliases, combo, apiKeys, ustawienia, ceny,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +## 3) Persistence Layer -Trwałość użytkowania: +Primary state DB (SQLite): -- fasada: `src/lib/usageDb.ts` (rozłożone moduły w `src/lib/usage/*`) -- Tabele SQLite w `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` -- opcjonalne artefakty plików pozostają w celu zapewnienia zgodności/debugowania (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- starsze pliki JSON są migrowane do SQLite poprzez migracje startowe, jeśli są obecne +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -Baza danych stanu domeny (SQLite): +Usage persistence: -- `src/lib/db/domainState.ts` — operacje CRUD na stanie domeny -- Tabele (utworzone w `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- Wzór pamięci podręcznej zapisu: mapy w pamięci są wiarygodne w czasie wykonywania; mutacje są zapisywane synchronicznie do SQLite; stan jest przywracany z bazy danych przy zimnym starcie## 4) Auth + Security Surfaces +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- Uwierzytelnianie plików cookie w panelu kontrolnym: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- Generowanie/weryfikacja klucza API: `src/shared/utils/apiKey.ts` -- Wpisy tajne dostawcy zostały zachowane we wpisach `providerConnections` -- Obsługa wychodzącego proxy poprzez `open-sse/utils/proxyFetch.ts` (env vars) i `open-sse/utils/networkProxy.ts` (konfigurowalne dla każdego dostawcy lub globalne)## 5) Cloud Sync +Domain State DB (SQLite): -- Inicjacja harmonogramu: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Zadanie okresowe: `src/shared/services/cloudSyncScheduler.ts` -- Zadanie okresowe: `src/shared/services/modelSyncScheduler.ts` -- Trasa kontrolna: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Decyzje awaryjne są podejmowane przez plik `open-sse/services/accountFallback.ts` przy użyciu kodów stanu i heurystyki komunikatów o błędach. Routing kombinowany dodaje jedną dodatkową osłonę: błędy 400 o zasięgu dostawcy, takie jak błędy blokowania zawartości nadrzędnego i sprawdzania poprawności ról, są traktowane jako awarie lokalne modelu, dzięki czemu późniejsze cele kombinacji mogą nadal działać.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -Odświeżanie podczas ruchu na żywo jest wykonywane wewnątrz `open-sse/handlers/chatCore.ts` poprzez executor `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -Okresowa synchronizacja jest wyzwalana przez „CloudSyncScheduler”, gdy włączona jest chmura.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -Pliki pamięci fizycznej: +Physical storage files: -- podstawowy DB środowiska wykonawczego: `${DATA_DIR}/storage.sqlite` -- żądanie wierszy dziennika: `${DATA_DIR}/log.txt` (artefakt zgodności/debugowania) -- archiwa ładunków wywołań strukturalnych: `${DATA_DIR}/call_logs/` -- opcjonalne sesje debugowania tłumacza/żądania: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: API zgodności -- `src/app/api/v1/providers/[dostawca]/*`: dedykowane trasy dla poszczególnych dostawców (czat, osadzanie, obrazy) -- `src/app/api/providers*`: dostawca CRUD, walidacja, testowanie -- `src/app/api/provider-nodes*`: niestandardowe zarządzanie kompatybilnymi węzłami -- `src/app/api/provider-models`: zarządzanie modelami niestandardowymi (CRUD) -- `src/app/api/models/route.ts`: API katalogu modeli (aliasy + modele niestandardowe) -- `src/app/api/oauth/*`: przepływy OAuth/kodu urządzenia -- `src/app/api/keys*`: cykl życia lokalnego klucza API -- `src/app/api/models/alias`: zarządzanie aliasami -- `src/app/api/combos*`: zarządzanie rezerwowymi kombinacjami -- `src/app/api/pricing`: zastąpienie cen przy kalkulacji kosztów -- `src/app/api/settings/proxy`: konfiguracja proxy (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: test połączenia wychodzącego proxy (POST) -- `src/app/api/usage/*`: API użycia i dzienników -- `src/app/api/sync/*` + `src/app/api/cloud/*`: synchronizacja z chmurą i pomocnicy obsługujący chmurę -- `src/app/api/cli-tools/*`: lokalni autorzy/weryfikatorzy konfiguracji CLI -- `src/app/api/settings/ip-filter`: lista dozwolonych/lista blokowanych adresów IP (GET/PUT) -- `src/app/api/settings/thinking-budget`: konfiguracja budżetu tokena myślącego (GET/PUT) -- `src/app/api/settings/system-prompt`: globalny monit systemowy (GET/PUT) -- `src/app/api/sessions`: lista aktywnych sesji (GET) -- `src/app/api/rate-limits`: stan limitu stawki na konto (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: analiza żądań, obsługa kombinacji, pętla wyboru konta -- `open-sse/handlers/chatCore.ts`: tłumaczenie, wysyłanie executora, obsługa ponawiania/odświeżania, konfiguracja strumienia -- `open-sse/executors/*`: zachowanie sieci i formatu specyficzne dla dostawcy### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: rejestracja i orkiestracja tłumaczy -- Poproś o tłumaczy: `open-sse/translator/request/*` -- Tłumacze odpowiedzi: `open-sse/translator/response/*` -- Stałe formatu: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: trwała konfiguracja/stan i trwałość domeny w SQLite -- `src/lib/localDb.ts`: reeksport kompatybilności dla modułów DB -- `src/lib/usageDb.ts`: fasada historii użytkowania/dzienników połączeń na tabelach SQLite## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Każdy dostawca ma wyspecjalizowany executor rozszerzający `BaseExecutor` (w `open-sse/executors/base.ts`), który zapewnia tworzenie adresów URL, konstruowanie nagłówków, ponawianie prób z wykładniczym wycofywaniem, przechwytywanie odświeżania poświadczeń i metodę orkiestracji `execute()`. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Wykonawca | Dostawca(-y) | Specjalna obsługa | -| -------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------- | -| `Domyślny wykonawca` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Razem, Fajerwerki, Cerebras, Cohere, NVIDIA | Dynamiczna konfiguracja adresu URL/nagłówka dla każdego dostawcy | -| `Egzekutor antygrawitacji` | Google Antygrawitacja | Niestandardowe identyfikatory projektów/sesji, ponowna próba po przeanalizowaniu | -| `Egzekutor Kodeksu` | Kodeks OpenAI | Wstrzykuje instrukcje systemowe, wymusza wysiłek rozumowania | -| `Egzekutor kursora` | Kursor IDE | Protokół ConnectRPC, kodowanie Protobuf, podpisywanie żądań poprzez sumę kontrolną | -| `GithubExecutor` | Drugi pilot GitHuba | Odświeżanie tokenu drugiego pilota, nagłówki naśladujące VSCode | -| `KiroExecutor` | Zaklinacz kodów AWS/Kiro | Format binarny AWS EventStream → Konwersja SSE | -| `GeminiCLIEExecutor` | Bliźnięta CLI | Cykl odświeżania tokena Google OAuth | +### Persistence -Wszyscy pozostali dostawcy (w tym niestandardowe kompatybilne węzły) używają `DefaultExecutor`.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Dostawca | Formatuj | Autoryzacja | Strumień | Non-Stream | Odświeżenie tokena | Korzystanie z interfejsu API | -| ------------------- | ----------------- | ----------------------------- | ----------------------- | ---------- | ------------------ | ---------------------------- | ------------------------------ | -| Klaudiusz | klaudia | Klucz API / OAuth | ✅ | ✅ | ✅ | ⚠️ Tylko administrator | -| Bliźnięta | Bliźnięta | Klucz API / OAuth | ✅ | ✅ | ✅ | ⚠️ Konsola chmurowa | -| Bliźnięta CLI | bliźnięta-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Konsola chmurowa | -| Antygrawitacja | antygrawitacja | OAuth | ✅ | ✅ | ✅ | ✅ Pełny limit API | -| OpenAI | otwieram | Klucz API | ✅ | ✅ | ❌ | ❌ | -| Kodeks | odpowiedzi openai | OAuth | ✅ zmuszony | ❌ | ✅ | ✅ Limity stawek | -| Drugi pilot GitHuba | otwieram | OAuth + token drugiego pilota | ✅ | ✅ | ✅ | ✅ Migawki kwot | -| Kursor | kursor | Niestandardowa suma kontrolna | ✅ | ✅ | ❌ | ❌ | -| Kiro | Kiro | AWS SSO OIDC | ✅ (Strumień zdarzenia) | ❌ | ✅ | ✅ Limity użytkowania | -| Qwen | otwieram | OAuth | ✅ | ✅ | ✅ | ⚠️ Na żądanie | -| Qoder | otwieram | OAuth (podstawowy) | ✅ | ✅ | ✅ | ⚠️ Na żądanie | -| OtwórzRouter | otwieram | Klucz API | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | klaudia | Klucz API | ✅ | ✅ | ❌ | ❌ | -| DeepSeek | otwieram | Klucz API | ✅ | ✅ | ❌ | ❌ | -| Groq | otwieram | Klucz API | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | otwieram | Klucz API | ✅ | ✅ | ❌ | ❌ | -| Mistral | otwieram | Klucz API | ✅ | ✅ | ❌ | ❌ | -| Zakłopotanie | otwieram | Klucz API | ✅ | ✅ | ❌ | ❌ | -| Razem AI | otwieram | Klucz API | ✅ | ✅ | ❌ | ❌ | -| Fajerwerki AI | otwieram | Klucz API | ✅ | ✅ | ❌ | ❌ | -| Cerebra | otwieram | Klucz API | ✅ | ✅ | ❌ | ❌ | -| Spójne | otwieram | Klucz API | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | otwieram | Klucz API | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Wykryte formaty źródłowe obejmują: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. -- "openai". -- „odpowiedzi openai”. -- ,,klaudia". -- ,,bliźnięta". +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | -Formaty docelowe obejmują: +All other providers (including custom compatible nodes) use the `DefaultExecutor`. -- Czat/odpowiedzi OpenAI -- Klaudiusz -- Koperta Gemini/Gemini-CLI/Antygrawitacyjna +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: + +- `openai` +- `openai-responses` +- `claude` +- `gemini` + +Target formats include: + +- OpenAI chat/Responses +- Claude +- Gemini/Gemini-CLI/Antigravity envelope - Kiro -- Kursor +- Cursor -Tłumaczenia używają**OpenAI jako formatu centralnego**— wszystkie konwersje przechodzą przez OpenAI jako pośredni:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Tłumaczenia są wybierane dynamicznie na podstawie kształtu ładunku źródłowego i formatu docelowego dostawcy. +Additional processing layers in the translation pipeline: -Dodatkowe warstwy przetwarzania w potoku tłumaczenia: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Oczyszczanie odpowiedzi**— Usuwa niestandardowe pola z odpowiedzi w formacie OpenAI (zarówno przesyłanych strumieniowo, jak i nie przesyłanych strumieniowo), aby zapewnić ścisłą zgodność z SDK --**Normalizacja ról**— Konwertuje „programista” → „system” dla celów innych niż OpenAI; łączy `system` → `użytkownik` dla modeli odrzucających rolę systemową (GLM, ERNIE) --**Pomyśl o ekstrakcji tagów**— Analizuje bloki `...` z treści do pola `reasoning_content` --**Ustrukturyzowane dane wyjściowe**— Konwertuje OpenAI `response_format.json_schema` na `responseMimeType` + `responseSchema` firmy Gemini## Supported API Endpoints +## Supported API Endpoints -| Punkt końcowy | Formatuj | Opiekun | -| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------ | -| `POST /v1/chat/uzupełnienia` | Czat OpenAI | `src/sse/handlers/chat.ts` | -| `POST /v1/wiadomości` | Wiadomości Claude'a | Ten sam program obsługi (wykryty automatycznie) | -| `POST /v1/odpowiedzi` | Odpowiedzi OpenAI | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/osadzania` | Osadzania OpenAI | `open-sse/handlers/embeddings.ts` | -| `POBIERZ /v1/osadzania` | Lista modeli | Trasa API | -| `POST /v1/images/generacje` | Obrazy OpenAI | `open-sse/handlers/imageGeneration.ts` | -| `POBIERZ /v1/obrazy/generacje` | Lista modeli | Trasa API | -| `POST /v1/providers/{provider}/chat/completions` | Czat OpenAI | Dedykowany dla każdego dostawcy z walidacją modelu | -| `POST /v1/providers/{provider}/embeddings` | Osadzania OpenAI | Dedykowany dla każdego dostawcy z walidacją modelu | -| `POST /v1/providers/{provider}/images/generations` | Obrazy OpenAI | Dedykowany dla każdego dostawcy z walidacją modelu | -| `POST /v1/messages/count_tokens` | Claude Liczba żetonów | Trasa API | -| `POBIERZ /v1/modele` | Lista modeli OpenAI | Ścieżka API (czat + osadzanie + obraz + modele niestandardowe) | -| `POBIERZ /api/models/catalog` | Katalog | Wszystkie modele pogrupowane według dostawcy + typu | -| `POST /v1beta/models/*:streamGenerateContent` | Pochodzący z Bliźniąt | Trasa API | -| `POBIERZ/PUT/USUŃ /api/settings/proxy` | Konfiguracja proxy | Konfiguracja serwera proxy sieci | -| `POST /api/settings/proxy/test` | Łączność proxy | Punkt końcowy testu kondycji/łączności serwera proxy | -| `POBIERZ/POST/USUŃ /api/provider-models` | Modele dostawców | Metadane modelu dostawcy obsługujące niestandardowe i zarządzane dostępne modele |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -Procedura obsługi obejścia (`open-sse/utils/bypassHandler.ts`) przechwytuje znane żądania „wyrzucenia” z Claude CLI — pingi rozgrzewające, wyodrębnianie tytułów i zliczanie tokenów — i zwraca**fałszywą odpowiedź**bez zużywania tokenów dostawcy nadrzędnego. Jest to wyzwalane tylko wtedy, gdy `User-Agent` zawiera `claude-cli`.## Request Logger Pipeline +## Bypass Handler -Rejestrator żądań (`open-sse/utils/requestLogger.ts`) zapewnia 7-etapowy potok rejestrowania debugowania, domyślnie wyłączony, włączony poprzez `ENABLE_REQUEST_LOGS=true`:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Pliki są zapisywane w `/logs//` dla każdej sesji żądania.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- czas oczekiwania na konto dostawcy w przypadku błędów przejściowych/szybkości/auth -- rezerwowe konto przed nieudanym żądaniem -- powrót do modelu kombi, gdy bieżąca ścieżka modelu/dostawcy zostanie wyczerpana## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- wstępne sprawdzenie i odświeżenie z ponowną próbą dla dostawców z możliwością odświeżania -- Ponowna próba 401/403 po próbie odświeżenia w ścieżce podstawowej## 3) Stream Safety +## 2) Token Expiry -- kontroler strumienia obsługujący rozłączenie -- strumień tłumaczeń z opróżnianiem na końcu strumienia i obsługą `[DONE]` -- rezerwowe oszacowanie użycia w przypadku braku metadanych dotyczących użycia dostawcy## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- pojawiają się błędy synchronizacji, ale lokalne środowisko wykonawcze trwa -- harmonogram ma logikę umożliwiającą ponawianie prób, ale wykonywanie okresowe obecnie domyślnie wywołuje synchronizację przy pojedynczej próbie## 5) Data Integrity +## 3) Stream Safety -- Migracje schematu SQLite i zaczepy do automatycznej aktualizacji przy uruchomieniu -- starsza ścieżka zgodności migracji JSON → SQLite## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Źródła widoczności w czasie wykonywania: +## 4) Cloud Sync Degradation -- logi konsoli z `src/sse/utils/logger.ts` -- agregacje użycia na żądanie w SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- czteroetapowe szczegółowe przechwytywanie ładunku w SQLite (`request_detail_logs`) gdy `settings.detailed_logs_enabled=true` -- logowanie statusu żądania tekstowego w `log.txt` (opcjonalnie/kompatybilne) -- opcjonalne głębokie logi żądań/tłumaczeń w `logs/`, gdy `ENABLE_REQUEST_LOGS=true` -- punkty końcowe użycia dashboardu (`/api/usage/*`) do wykorzystania interfejsu użytkownika +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -Szczegółowe przechwytywanie ładunku żądania przechowuje do czterech etapów ładunku JSON na kierowane połączenie: +## 5) Data Integrity -- surowe żądanie otrzymane od klienta -- przetłumaczone żądanie faktycznie wysłane w górę -- odpowiedź dostawcy zrekonstruowana jako JSON; Odpowiedzi przesyłane strumieniowo są kompresowane do końcowego podsumowania plus metadane strumienia -- ostateczna odpowiedź klienta zwrócona przez OmniRoute; odpowiedzi przesyłane strumieniowo są przechowywane w tej samej zwartej formie podsumowania## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- Sekret JWT („JWT_SECRET”) zabezpiecza weryfikację/podpisywanie plików cookie sesji panelu kontrolnego -- Początkowe ładowanie hasła (`INITIAL_PASSWORD`) powinno być jawnie skonfigurowane na potrzeby udostępniania przy pierwszym uruchomieniu -- Klucz API Sekret HMAC (`API_KEY_SECRET`) zabezpiecza wygenerowany lokalny format klucza API -- Sekrety dostawcy (klucze/tokeny API) są zachowywane w lokalnej bazie danych i powinny być chronione na poziomie systemu plików -- Punkty końcowe synchronizacji w chmurze opierają się na uwierzytelnianiu klucza API + semantyce identyfikatora maszyny## Environment and Runtime Matrix +## Observability and Operational Signals -Zmienne środowiskowe aktywnie używane przez kod: +Runtime visibility sources: -- Aplikacja/autoryzacja: `JWT_SECRET`, `INITIAL_PASSWORD` -- Pamięć: `DATA_DIR` -- Zgodne zachowanie węzła: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Opcjonalne zastąpienie bazy pamięci (Linux/macOS, gdy `DATA_DIR` nie jest ustawione): `XDG_CONFIG_HOME` -- Haszowanie zabezpieczeń: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Rejestrowanie: `ENABLE_REQUEST_LOGS` -- Adres URL synchronizacji/chmury: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- Wychodzący serwer proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` i warianty z małymi literami -- Flagi funkcji SOCKS5: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- Pomocnicy platformy/środowiska wykonawczego (konfiguracja nie specyficzna dla aplikacji): `APPDATA`, `NODE_ENV`, `PORT`, `NAZWA HOSTA`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` i `localDb` mają tę samą podstawową politykę katalogową (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) z migracją starszych plików. -2. `/api/v1/route.ts` deleguje do tego samego ujednoliconego narzędzia do tworzenia katalogów, którego używa `/api/v1/models` (`src/app/api/v1/models/catalog.ts`), aby uniknąć dryfu semantycznego. -3. Rejestrator żądań zapisuje pełne nagłówki/treść, gdy jest włączony; traktuj katalog dzienników jako poufny. -4. Zachowanie chmury zależy od prawidłowego `NEXT_PUBLIC_BASE_URL` i osiągalności punktu końcowego chmury. -5. Katalog `open-sse/` jest publikowany jako pakiet `@omniroute/open-sse`**npm workspace**. Kod źródłowy importuje go poprzez `@omniroute/open-sse/...` (rozwiązany przez Next.js `transpilePackages`). Ścieżki plików w tym dokumencie nadal używają nazwy katalogu `open-sse/` dla zachowania spójności. -6. Wykresy na pulpicie nawigacyjnym korzystają z**Recharts**(oparte na SVG) w celu uzyskania przystępnych, interaktywnych wizualizacji analitycznych (wykresy słupkowe wykorzystania modelu, tabele podziału dostawców ze wskaźnikami sukcesu). -7. Testy E2E wykorzystują**Playwright**(`tests/e2e/`), uruchamiają się poprzez `npm run test:e2e`. Testy jednostkowe wykorzystują**program uruchamiający testy Node.js**(`tests/unit/`), uruchamiane poprzez `npm run test:unit`. Kod źródłowy pod `src/` to**TypeScript**(`.ts`/`.tsx`); obszarem roboczym `open-sse/` pozostaje JavaScript (`.js`). -8. Strona ustawień jest podzielona na 5 zakładek: Bezpieczeństwo, Routing (6 globalnych strategii: najpierw wypełnij, okrężnie, p2c, losowa, najrzadziej używana, zoptymalizowana pod względem kosztów), Odporność (edytowalne limity szybkości, wyłącznik automatyczny, zasady), AI (przemyślany budżet, monit systemowy, pamięć podręczna podpowiedzi), Zaawansowane (proxy).## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- Kompiluj ze źródła: `npm run build` -- Zbuduj obraz Dockera: `docker build -t omniroute.` -- Uruchom usługę i sprawdź: -- `POBIERZ /api/ustawienia` -- `POBIERZ /api/v1/models` -- Podstawowy docelowy adres URL CLI powinien mieć postać `http://:20128/v1`, gdy `PORT=20128` +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` +- `GET /api/v1/models` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/pl/docs/FEATURES.md b/docs/i18n/pl/docs/FEATURES.md index 03a149a90a..85b7308975 100644 --- a/docs/i18n/pl/docs/FEATURES.md +++ b/docs/i18n/pl/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Wizualny przewodnik po każdej sekcji pulpitu nawigacyjnego OmniRoute.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Zarządzaj połączeniami dostawców AI: dostawcami OAuth (Claude Code, Codex, Gemini CLI), dostawcami kluczy API (Groq, DeepSeek, OpenRouter) i bezpłatnymi dostawcami (Qoder, Qwen, Kiro). Konta Kiro umożliwiają śledzenie salda kredytu — pozostałe środki, całkowity limit i datę odnowienia widoczne w Panelu → Wykorzystanie.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Twórz kombinacje routingu modeli z 6 strategiami: priorytetową, ważoną, okrężną, losową, najrzadziej używaną i zoptymalizowaną pod względem kosztów. Każda kombinacja łączy wiele modeli z automatycznym przywracaniem i zawiera szybkie szablony oraz kontrole gotowości.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Kompleksowa analiza użytkowania obejmująca zużycie tokenów, szacunki kosztów, mapy cieplne aktywności, tygodniowe wykresy dystrybucji i zestawienia poszczególnych dostawców.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Monitorowanie w czasie rzeczywistym: czas pracy, pamięć, wersja, percentyle opóźnień (p50/p95/p99), statystyki pamięci podręcznej i stany wyłączników automatycznych dostawcy.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Cztery tryby debugowania tłumaczeń API:**Playground**(konwerter formatów),**Chat Tester**(żądania na żywo),**Test Bench**(testy wsadowe) i**Live Monitor**(strumień w czasie rzeczywistym).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Przetestuj dowolny model bezpośrednio z pulpitu nawigacyjnego. Wybierz dostawcę, model i punkt końcowy, pisz podpowiedzi za pomocą edytora Monaco, przesyłaj strumieniowo odpowiedzi w czasie rzeczywistym, przerwij transmisję w połowie i przeglądaj metryki czasu.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Konfigurowalne motywy kolorystyczne dla całego pulpitu nawigacyjnego. Wybierz jeden z 7 gotowych kolorów (koralowy, niebieski, czerwony, zielony, fioletowy, pomarańczowy, cyjan) lub utwórz własny motyw, wybierając dowolny kolor szesnastkowy. Obsługuje tryb jasny, ciemny i systemowy.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Rozbudowany panel ustawień z zakładkami: +Comprehensive settings panel with tabs: --**Ogólne**— Pamięć systemowa, zarządzanie kopiami zapasowymi (baza danych eksportu/importu) -**Wygląd**— Selektor motywu (ciemny/jasny/system), ustawienia motywu kolorów i kolory niestandardowe, widoczność dziennika stanu, elementy sterujące widocznością elementów na pasku bocznym -**Bezpieczeństwo**— ochrona punktu końcowego API, niestandardowe blokowanie dostawców, filtrowanie IP, informacje o sesji -**Routing**— Aliasy modeli, degradacja zadań w tle -**Odporność**— Utrzymywanie limitów szybkości, dostrajanie wyłączników automatycznych, automatyczne wyłączanie zbanowanych kont, monitorowanie wygaśnięcia ważności dostawcy -**Zaawansowane**— Zastąpienie konfiguracji, ścieżka audytu konfiguracji, awaryjny tryb degradacji![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Konfiguracja narzędzi do kodowania AI za pomocą jednego kliknięcia: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Kontynuuj, Cursor i Factory Droid. Zawiera automatyczne stosowanie/resetowanie konfiguracji, profile połączeń i mapowanie modeli.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Panel do wyszukiwania agentów CLI i zarządzania nimi. Pokazuje siatkę 14 wbudowanych agentów (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) z: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Stan instalacji**— Zainstalowano / Nie znaleziono z wykrywaniem wersji -**Identyfikatory protokołów**— stdio, HTTP itp. -**Niestandardowi agenci**— Zarejestruj dowolne narzędzie CLI za pomocą formularza (nazwa, plik binarny, polecenie wersji, argumenty spawnu) -**Dopasowanie odcisków palców CLI**— Przełącznik dla poszczególnych dostawców w celu dopasowania natywnych podpisów żądań CLI, zmniejszając ryzyko bana przy jednoczesnym zachowaniu adresu IP serwera proxy--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Generuj obrazy, filmy i muzykę z poziomu pulpitu nawigacyjnego. Obsługuje OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open i MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Rejestrowanie żądań w czasie rzeczywistym z filtrowaniem według dostawcy, modelu, konta i klucza API. Pokazuje kody stanu, użycie tokenu, opóźnienie i szczegóły odpowiedzi.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Twój ujednolicony punkt końcowy interfejsu API z podziałem możliwości: uzupełnianie czatu, interfejs API odpowiedzi, osadzanie, generowanie obrazu, zmiana rankingu, transkrypcja audio, zamiana tekstu na mowę, moderacje i zarejestrowane klucze API. Integracja z Cloudflare Quick Tunnel i obsługa proxy w chmurze dla zdalnego dostępu.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -Twórz, ustalaj zakres i unieważniaj klucze API. Każdy klucz może być ograniczony do określonych modeli/dostawców z pełnym dostępem lub uprawnieniami tylko do odczytu. Wizualne zarządzanie kluczami ze śledzeniem użycia.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Śledzenie działań administracyjnych z filtrowaniem według typu działania, aktora, celu, adresu IP i znacznika czasu. Pełna historia zdarzeń związanych z bezpieczeństwem.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Natywna aplikacja komputerowa Electron dla systemów Windows, macOS i Linux. Uruchom OmniRoute jako samodzielną aplikację z integracją z zasobnikiem systemowym, obsługą offline, automatyczną aktualizacją i instalacją jednym kliknięciem. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Kluczowe cechy: +Key features: -- Odpytywanie o gotowość serwera (bez pustego ekranu przy zimnym starcie) -- Taca systemowa z zarządzaniem portami -- Polityka bezpieczeństwa treści -- Blokada jednoinstancyjna -- Automatyczna aktualizacja po ponownym uruchomieniu -- Interfejs użytkownika zależny od platformy (sygnalizacja świetlna macOS, domyślny pasek tytułowy Windows/Linux) -- Ulepszone pakowanie kompilacji Electron — dowiązania symboliczne „node_modules” w samodzielnym pakiecie są wykrywane i odrzucane przed pakowaniem, zapobiegając uzależnieniu środowiska wykonawczego od maszyny budującej (wersja 2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Zobacz [`electron/README.md`](../electron/README.md), aby uzyskać pełną dokumentację. +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/pl/docs/TROUBLESHOOTING.md b/docs/i18n/pl/docs/TROUBLESHOOTING.md index 5b872a02de..c75630df0e 100644 --- a/docs/i18n/pl/docs/TROUBLESHOOTING.md +++ b/docs/i18n/pl/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -Typowe problemy i rozwiązania dla OmniRoute.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Problem | Rozwiązanie | -| ------------------------------------------ | -------------------------------------------------------------------------------------- | --- | -| Pierwsze logowanie nie działa | Ustaw `INITIAL_PASSWORD` w `.env` (brak wartości domyślnej) | -| Panel kontrolny otwiera się na złym porcie | Ustaw `PORT=20128` i `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Brak logów żądań w `logs/` | Ustaw `ENABLE_REQUEST_LOGS=true` | -| EACCES: odmowa pozwolenia | Ustaw `DATA_DIR=/ścieżka/do/zapisu/katalog`, aby zastąpić `~/.omniroute` | -| Strategia routingu nie jest zapisywana | Aktualizacja do wersji 1.4.11+ (poprawka schematu Zoda zapewniająca trwałość ustawień) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Przyczyna:**Wyczerpany limit dostawcy. +**Cause:** Provider quota exhausted. -**Poprawka:** +**Fix:** -1. Sprawdź moduł śledzenia limitów na pulpicie nawigacyjnym -2. Użyj kombinacji z poziomami rezerwowymi -3. Przejdź na tańszy/bezpłatny poziom### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Przyczyna:**Wyczerpany limit subskrypcji. +### Rate Limiting -**Poprawka:** +**Cause:** Subscription quota exhausted. -- Dodaj rezerwę: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Użyj GLM/MiniMax jako taniej kopii zapasowej### OAuth Token Expired +**Fix:** -OmniRoute automatycznie odświeża tokeny. Jeśli problemy nadal występują: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Panel kontrolny → Dostawca → Połącz ponownie -2. Usuń i ponownie dodaj połączenie dostawcy--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Sprawdź, czy `BASE_URL` wskazuje na działającą instancję (np. `http://localhost:20128`) -2. Sprawdź, czy `CLOUD_URL` wskazuje na punkt końcowy Twojej chmury (np. `https://omniroute.dev`) -3. Zachowaj wyrównanie wartości `NEXT_PUBLIC_*` z wartościami po stronie serwera### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Objaw:**`Nieoczekiwany token „d”...` na punkcie końcowym chmury dla połączeń innych niż przesyłanie strumieniowe. +### Cloud `stream=false` Returns 500 -**Przyczyna:**Upstream zwraca ładunek SSE, podczas gdy klient oczekuje JSON. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Rozwiązanie:**użyj parametru „stream=true” w przypadku bezpośrednich połączeń w chmurze. Lokalne środowisko wykonawcze obejmuje rezerwę SSE → JSON.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Utwórz nowy klucz z lokalnego pulpitu nawigacyjnego (`/api/keys`) -2. Uruchom synchronizację z chmurą: Włącz chmurę → Synchronizuj teraz -3. Stare/niezsynchronizowane klucze nadal mogą zwracać „401” w chmurze--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Sprawdź pola wykonawcze: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. W trybie przenośnym: użyj docelowego obrazu `runner-cli` (w pakiecie CLI) -3. W trybie montowania hosta: ustaw `CLI_EXTRA_PATHS` i zamontuj katalog bin hosta jako tylko do odczytu -4. Jeśli `installed=true` i `runnable=false`: znaleziono plik binarny, ale kontrola stanu nie powiodła się### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Sprawdź statystyki użytkowania w Panelu → Użycie -2. Zmień model podstawowy na GLM/MiniMax -3. Używaj bezpłatnej warstwy (Gemini CLI, Qoder) do zadań niekrytycznych -4. Ustaw budżety kosztów według klucza API: Panel → Klucze API → Budżet--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Ustaw `ENABLE_REQUEST_LOGS=true` w swoim pliku `.env`. Dzienniki pojawiają się w katalogu `logs/`.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Stan główny: `${DATA_DIR}/storage.sqlite` (dostawcy, kombinacje, aliasy, klucze, ustawienia) -- Użycie: tabele SQLite w `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + opcjonalnie `${DATA_DIR}/log.txt` i `${DATA_DIR}/call_logs/` -- Żądaj logów: `/logs/...` (gdy `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Gdy wyłącznik automatyczny dostawcy jest OTWARTY, żądania są blokowane do czasu upłynięcia czasu odnowienia. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Poprawka:** +**Fix:** -1. Przejdź do**Panel sterowania → Ustawienia → Odporność** -2. Sprawdź kartę wyłącznika dla odpowiedniego dostawcy -3. Kliknij**Resetuj wszystko**, aby wyczyścić wszystkie wyłączniki, lub poczekaj, aż upłynie czas odnowienia -4. Przed zresetowaniem sprawdź, czy dostawca jest rzeczywiście dostępny### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Jeśli dostawca wielokrotnie wchodzi w stan OPEN: +### Provider keeps tripping the circuit breaker -1. Sprawdź**Panel kontrolny → Kondycja → Kondycja dostawcy**pod kątem wzorca awarii -2. Przejdź do**Ustawienia → Odporność → Profile dostawców**i zwiększ próg awarii -3. Sprawdź, czy dostawca zmienił limity API lub wymaga ponownego uwierzytelnienia -4. Sprawdź dane telemetryczne dotyczące opóźnień — duże opóźnienia mogą powodować awarie wynikające z przekroczenia limitu czasu--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Upewnij się, że używasz prawidłowego przedrostka: `deepgram/nova-3` lub `assemblyai/best` -- Sprawdź, czy dostawca jest podłączony w**Panelu → Dostawcy**### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Sprawdź obsługiwane formaty audio: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- Sprawdź, czy rozmiar pliku mieści się w granicach dostawcy (zwykle < 25 MB) -- Sprawdź ważność klucza API dostawcy na karcie dostawcy--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Użyj**Panel kontrolny → Tłumacz**, aby debugować problemy z tłumaczeniem formatu: +Use **Dashboard → Translator** to debug format translation issues: -| Tryb | Kiedy stosować | -| ------------------------- | ---------------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Plac zabaw** | Porównaj formaty wejścia/wyjścia obok siebie — wklej nieudane żądanie, aby zobaczyć, jak zostanie przetłumaczone | -| **Tester czatu** | Wysyłaj wiadomości na żywo i sprawdzaj pełny ładunek żądania/odpowiedzi, w tym nagłówki | -| **Stolik testowy** | Przeprowadź testy wsadowe dla kombinacji formatów, aby dowiedzieć się, które tłumaczenia są uszkodzone | -| **Monitorowanie na żywo** | Obserwuj przepływ żądań w czasie rzeczywistym, aby wykryć sporadyczne problemy z tłumaczeniem | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Tagi myślenia nie pojawiają się**— Sprawdź, czy dostawca docelowy obsługuje myślenie i ustawienie budżetu na myślenie -**Porzucanie wywołań narzędzi**— Niektóre tłumaczenia formatów mogą usuwać nieobsługiwane pola; sprawdź w trybie placu zabaw -**Brak podpowiedzi systemowej**— Claude i Gemini inaczej obsługują podpowiedzi systemowe; sprawdź wynik tłumaczenia -**SDK zwraca nieprzetworzony ciąg znaków zamiast obiektu**— Naprawiono w wersji 1.1.0: narzędzie do czyszczenia odpowiedzi usuwa teraz niestandardowe pola (`x_groq`, `usage_breakdown` itp.), które powodują błędy sprawdzania poprawności OpenAI SDK Pydantic -**GLM/ERNIE odrzuca rolę „systemową”**— Naprawiono w wersji 1.1.0: normalizator ról automatycznie łączy komunikaty systemowe z komunikatami użytkownika w przypadku niekompatybilnych modeli -**Rola „programisty” nie została rozpoznana**— Naprawiono w wersji 1.1.0: automatycznie konwertowana na „system” dla dostawców innych niż OpenAI -**`json_schema` nie działa z Gemini**— Naprawiono w wersji 1.1.0: `response_format` jest teraz konwertowany na `responseMimeType` + `responseSchema` Gemini--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- Automatyczne ograniczenie szybkości dotyczy tylko dostawców kluczy API (nie OAuth/subskrypcja) -- Sprawdź, czy**Ustawienia → Odporność → Profile dostawców**ma włączone automatyczne ograniczenie stawek -- Sprawdź, czy dostawca zwraca kody stanu „429” lub nagłówki „Retry-After”.### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -Profile dostawców obsługują następujące ustawienia: +### Tuning exponential backoff --**Opóźnienie bazowe**— Początkowy czas oczekiwania po pierwszej awarii (domyślnie: 1 s) -**Maks. opóźnienie**— Maksymalny limit czasu oczekiwania (domyślnie: 30 s) -**Mnożnik**— O ile zwiększyć opóźnienie przy kolejnej awarii (domyślnie: 2x)### Anti-thundering herd +Provider profiles support these settings: -Gdy wiele jednoczesnych żądań trafia do dostawcy z ograniczoną szybkością, OmniRoute używa mutexu i automatycznego ograniczania szybkości, aby serializować żądania i zapobiegać kaskadowym błędom. Jest to automatyczne w przypadku dostawców kluczy API.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Niektórzy użytkownicy OmniRoute umieszczają bramę przed stosami RAG lub agentów. W takich konfiguracjach często można zaobserwować dziwny wzór: OmniRoute wygląda na w porządku (dostawcy działają, profile routingu w porządku, brak alertów o limitach szybkości), ale ostateczna odpowiedź nadal jest błędna. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -W praktyce zdarzenia te zwykle mają miejsce w dalszym rurociągu RAG, a nie w samej bramie. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Jeśli chcesz, aby wspólne słownictwo opisało te awarie, możesz użyć WFGY ProblemMap, zewnętrznego zasobu tekstowego licencji MIT, który definiuje szesnaście powtarzających się wzorców awarii RAG/LLM. Na wysokim poziomie obejmuje: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- dryf wyszukiwania i zerwanie granic kontekstu -- puste lub nieaktualne indeksy i magazyny wektorów -- osadzanie a niedopasowanie semantyczne -- szybkie problemy z montażem i oknem kontekstowym -- załamanie logiki i zbytnia pewność siebie w odpowiedziach -- błędy w długim łańcuchu i koordynacji agentów -- pamięć wieloagentowa i dryf ról -- problemy z wdrażaniem i porządkowaniem ładowania początkowego +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -Pomysł jest prosty: +The idea is simple: -1. Kiedy sprawdzasz złą reakcję, przechwyć: - - zadanie i żądanie użytkownika - - kombinacja tras lub dostawców w OmniRoute - - dowolny kontekst RAG używany w dalszej części procesu (pobrane dokumenty, wywołania narzędzi itp.) -2. Przypisz incydent do jednego lub dwóch numerów WFGY ProblemMap („Nr 1” … „Nr 16”). -3. Zapisz numer na swoim własnym pulpicie nawigacyjnym, elemencie Runbook lub narzędziu do śledzenia zdarzeń obok dzienników OmniRoute. -4. Użyj odpowiedniej strony WFGY, aby zdecydować, czy musisz zmienić stos RAG, aporter lub strategię routingu. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -Pełny tekst i konkretne przepisy znajdują się tutaj (licencja MIT, tylko tekst): +Full text and concrete recipes live here (MIT license, text only): -[Plik README ProblemMap WFGY](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) +[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -Możesz zignorować tę sekcję, jeśli nie uruchamiasz potoków RAG ani agentów za OmniRoute.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**Problemy z GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Architektura**: Zobacz [`docs/ARCHITECTURE.md`](ARCHITECTURE.md), aby uzyskać szczegółowe informacje wewnętrzne -**Dokumentacja API**: Zobacz [`docs/API_REFERENCE.md`](API_REFERENCE.md) dla wszystkich punktów końcowych -**Panel stanu**: Sprawdź**Panel kontrolny → Zdrowie**, aby sprawdzić stan systemu w czasie rzeczywistym -**Tłumacz**: Użyj**Panel kontrolny → Tłumacz**, aby debugować problemy z formatem +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/pl/llm.txt b/docs/i18n/pl/llm.txt new file mode 100644 index 0000000000..076b50b4b1 --- /dev/null +++ b/docs/i18n/pl/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Polski) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Przegląd + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Bezpieczeństwo +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/pt-BR/README.md b/docs/i18n/pt-BR/README.md index e50d70d229..b3ee1b033f 100644 --- a/docs/i18n/pt-BR/README.md +++ b/docs/i18n/pt-BR/README.md @@ -4,6 +4,7 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. _Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ @@ -247,7 +248,7 @@ Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Eve - **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention - **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI - **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next -- **Custom Combos** — Customizable fallback chains with 9 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random) +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) - **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard @@ -1308,7 +1309,17 @@ Then in `/dashboard/media` → **Transcription** tab: upload any audio or video ## 💡 Key Features -OmniRoute v2.0 is built as an operational platform, not just a relay proxy. +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. + +### 🆕 New — v3.5.5 Highlights (Apr 2026) + +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | ### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) @@ -1360,7 +1371,8 @@ OmniRoute v2.0 is built as an operational platform, not just a relay proxy. | 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | | 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | | 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | -| 🎨 **Custom Combos** | 9 balancing strategies + fallback chain control | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | | 🌐 **Wildcard Router** | `provider/*` dynamic routing | | 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | | 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | @@ -2187,9 +2199,10 @@ Se não quiser criar credenciais próprias agora, ainda é possível usar o flux | ---------------------------------------------- | --------------------------------------------------- | | [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | | [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | -| [MCP Server](open-sse/mcp-server/README.md) | 16 MCP tools, IDE configs, Python/TS/Go clients | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | | [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | | [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | | [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | | [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | | [Contributing](CONTRIBUTING.md) | Development setup and guidelines | diff --git a/docs/i18n/pt-BR/docs/ARCHITECTURE.md b/docs/i18n/pt-BR/docs/ARCHITECTURE.md index 77b889d975..d8e29eb828 100644 --- a/docs/i18n/pt-BR/docs/ARCHITECTURE.md +++ b/docs/i18n/pt-BR/docs/ARCHITECTURE.md @@ -4,6 +4,8 @@ --- + + _Last updated: 2026-03-28_ ## Executive Summary @@ -36,6 +38,7 @@ Core capabilities: - Anti-thundering herd protection with mutex locking - Signature-based request deduplication cache - Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity - Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) - Policy engine for centralized request evaluation (lockout → budget → fallback) - Request telemetry with p50/p95/p99 latency aggregation @@ -222,6 +225,8 @@ Services (business logic): - Wildcard model routing: `open-sse/services/wildcardRouter.ts` - Rate limit management: `open-sse/services/rateLimitManager.ts` - Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions Domain layer modules: @@ -802,7 +807,10 @@ Environment variables actively used by code: 5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. 6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). 7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). -8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. ## Operational Verification Checklist diff --git a/docs/i18n/pt-BR/docs/FEATURES.md b/docs/i18n/pt-BR/docs/FEATURES.md index 029ce37d98..3d12e84ff6 100644 --- a/docs/i18n/pt-BR/docs/FEATURES.md +++ b/docs/i18n/pt-BR/docs/FEATURES.md @@ -4,6 +4,8 @@ --- + + Visual guide to every section of the OmniRoute dashboard. --- @@ -18,7 +20,7 @@ Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI) ## 🎨 Combos -Create model routing combos with 6 strategies: priority, weighted, round-robin, random, least-used, and cost-optimized. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. ![Combos Dashboard](screenshots/02-combos.png) @@ -68,7 +70,7 @@ Comprehensive settings panel with tabs: - **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls - **Security** — API endpoint protection, custom provider blocking, IP filtering, session info - **Routing** — Model aliases, background task degradation -- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration - **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode ![Settings Dashboard](screenshots/06-settings.png) @@ -94,6 +96,30 @@ Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in a --- +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- + ## 🖼️ Media _(v2.0.3+)_ Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. diff --git a/docs/i18n/pt-BR/docs/TROUBLESHOOTING.md b/docs/i18n/pt-BR/docs/TROUBLESHOOTING.md index dd8797cde1..b627c270ef 100644 --- a/docs/i18n/pt-BR/docs/TROUBLESHOOTING.md +++ b/docs/i18n/pt-BR/docs/TROUBLESHOOTING.md @@ -4,6 +4,8 @@ --- + + Common problems and solutions for OmniRoute. --- @@ -17,6 +19,60 @@ Common problems and solutions for OmniRoute. | No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | | EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | | Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. --- diff --git a/docs/i18n/pt-BR/llm.txt b/docs/i18n/pt-BR/llm.txt new file mode 100644 index 0000000000..cd4299d1c4 --- /dev/null +++ b/docs/i18n/pt-BR/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Português (Brasil)) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Visão Geral + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Segurança +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/pt/README.md b/docs/i18n/pt/README.md index 4acb48e1ab..11d72c1eb3 100644 --- a/docs/i18n/pt/README.md +++ b/docs/i18n/pt/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Seu proxy de API universal — um endpoint, mais de 60 provedores, zero tempo de inatividade. Agora com**Servidor MCP (25 ferramentas)**,**Protocolo A2A**,**Sistemas de memória/habilidades**e**Aplicativo Electron Desktop**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Conclusões de bate-papo • Incorporações • Geração de imagens • Vídeo • Música • Áudio • Reclassificação •**Pesquisa na Web**• Servidor MCP • Protocolo A2A • 100% TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Seu proxy de API universal — um endpoint, mais de 60 provedores, zero tempo d [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Site](https://omniroute.online) • [🚀 Início rápido](#-início rápido) • [💡 Recursos](#-recursos-chave) • [📖 Documentos](#-documentação) • [💰 Preços](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Disponível em:**🇺🇸 [Inglês](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Francês](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Alemão](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magiar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonésia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Holanda](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polês](docs/i18n/pl/README.md) | 🇸🇰 [Eslovênia](docs/i18n/sk/README.md) | 🇸🇪 [Suécia](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,553 +60,629 @@ _Seu proxy de API universal — um endpoint, mais de 60 provedores, zero tempo d ## 📸 Dashboard Preview - -Clique para ver capturas de tela do painel +
+Click to see dashboard screenshots -| Página | Captura de tela | -| -------------------- | ----------------------------------------------------- | ---------- | -| **Fornecedores** | ![Provedores](docs/screenshots/01-providers.png) | -| **Combos** | ![Combos](docs/screenshots/02-combos.png) | -| **Análise** | ![Analytics](docs/screenshots/03-analytics.png) | -| **Saúde** | ![Saúde](docs/screenshots/04-health.png) | -| **Tradutor** | ![Tradutor](docs/screenshots/05-translator.png) | -| **Configurações** | ![Configurações](docs/screenshots/06-settings.png) | -| **Ferramentas CLI** | ![Ferramentas CLI](docs/screenshots/07-cli-tools.png) | -| **Registros de uso** | ![Uso](docs/screenshots/08-usage.png) | -| **Pontos finais** | ![Pontos finais](docs/screenshots/09-endpoint.png) |
| +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | + + --- ### 🤖 Free AI Provider for your favorite coding agents -_Conecte qualquer ferramenta IDE ou CLI com tecnologia de IA por meio do OmniRoute - gateway de API gratuito para codificação ilimitada._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ - + - - - - - - - - - - - +
+ OpenClaw
OpenClaw

- ⭐ 205 mil + ⭐ 205K
+ NanoBot
NanoBot

- ⭐ 20,9K + ⭐ 20.9K
+ PicoClaw
- Pico Garra + PicoClaw

- ⭐ 14,6K + ⭐ 14.6K
+ ZeroClaw
ZeroClaw

- ⭐ 9,9K + ⭐ 9.9K
+ IronClaw
- Garra de Ferro + IronClaw

- ⭐ 2,1K + ⭐ 2.1K
+ OpenCode
OpenCode

⭐ 106K
+ Codex CLI
- CLI do Codex + Codex CLI

- ⭐ 60,8K + ⭐ 60.8K
+ - Código Claude
- Código Claude + Claude Code
+ Claude Code

- ⭐ 67,3K + ⭐ 67.3K
+ Gemini CLI
- CLI Gemini + Gemini CLI

- ⭐ 94,7K + ⭐ 94.7K
+ - Código Quilo
- Código do quilo + Kilo Code
+ Kilo Code

- ⭐ 15,5 mil + ⭐ 15.5K
-📡 Todos os agentes se conectam via http://localhost:20128/v1 ou http://cloud.omniroute.online/v1 — uma configuração, modelos e cota ilimitados--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Pare de desperdiçar dinheiro e atingir limites:** +**Stop wasting money and hitting limits:** -- A cota de assinatura expira sem ser utilizada todos os meses -- Os limites de taxa impedem você de codificar no meio -- APIs caras (US$ 20-50/mês por provedor) -- Troca manual entre provedores +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute resolve isso:** +**OmniRoute solves this:** -- ✅**Maximize as assinaturas**- Rastreie a cota, use cada bit antes de redefinir -- ✅**Fullback automático**- Assinatura → Chave de API → Barato → Gratuito, tempo de inatividade zero -- ✅**Múltiplas contas**- Round-robin entre contas por provedor -- ✅**Universal**- Funciona com Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, qualquer ferramenta CLI--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Junte-se à nossa comunidade!**[Grupo WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Obtenha ajuda, compartilhe dicas e fique atualizado. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Site**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Problemas**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Grupo comunitário](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Contribuindo**: Consulte [CONTRIBUTING.md](CONTRIBUTING.md), abra um PR ou escolha uma `boa primeira edição` -**Projeto Original**: [9router by decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Ao abrir um problema, execute o comando system-info e anexe o arquivo gerado:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Isso gera um `system-info.txt` com sua versão do Node.js, versão do OmniRoute, detalhes do sistema operacional, ferramentas CLI instaladas (qoder, gemini, claude, codex, antigravity, droid, etc.), status do Docker/PM2 e pacotes do sistema - tudo o que precisamos para reproduzir seu problema rapidamente. Anexe o arquivo diretamente ao seu problema do GitHub.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Todo desenvolvedor que usa ferramentas de IA enfrenta esses problemas diariamente.**O OmniRoute foi criado para resolver todos eles, desde custos excessivos até bloqueios regionais, desde fluxos quebrados de OAuth até operações de protocolo e observabilidade empresarial. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. - -💸 1. "Eu pago por uma assinatura cara, mas ainda sou interrompido por limites" +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Os desenvolvedores pagam US$ 20–200/mês pelo Claude Pro, Codex Pro ou GitHub Copilot. Mesmo pagando, a cota tem um limite máximo – 5h de uso, limites semanais ou limites de taxa por minuto. No meio da sessão de codificação, o provedor para de responder e o desenvolvedor perde fluxo e produtividade. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Como o OmniRoute resolve isso:** +**How OmniRoute solves it:** --**Smart 4-Tier Fallback**— Se a cota de assinatura acabar, redireciona automaticamente para API Key → Barato → Gratuito sem intervenção manual --**Rastreamento de limites do provedor**— Instantâneos de cota armazenados em cache são atualizados em uma programação do lado do servidor (padrão `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) com atualização manual disponível na UI --**Suporte para múltiplas contas**— Várias contas por provedor com round-robin automático — quando uma acabar, muda para a próxima --**Combos personalizados**— Cadeias alternativas personalizáveis com 9 estratégias de balanceamento (prioridade, ponderada, preencher primeiro, round-robin, P2C, aleatória, menos usada, com custo otimizado, estritamente aleatória) --**Codex Business Quotas**— Monitoramento de cotas de espaço de trabalho de negócios/equipe diretamente no painel
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard - -🔌 2. "Preciso usar vários provedores, mas cada um tem uma API diferente" + -OpenAI usa um formato, Claude (Anthropic) usa outro, Gemini ainda outro. Se um desenvolvedor quiser testar modelos de diferentes provedores ou fazer fallback entre eles, ele precisará reconfigurar SDKs, alterar endpoints e lidar com formatos incompatíveis. Provedores personalizados (FriendLI, NIM) possuem endpoints de modelo não padrão. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Como o OmniRoute resolve isso:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Endpoint unificado**— Um único `http://localhost:20128/v1` serve como proxy para todos os mais de 60 provedores --**Tradução de formato**— Automática e transparente: OpenAI ↔ Claude ↔ Gemini ↔ API de respostas --**Response Sanitization**— Remove campos não padrão (`x_groq`, `usage_breakdown`, `service_tier`) que quebram o OpenAI SDK v1.83+ --**Normalização de funções**— Converte `desenvolvedor` → `sistema` para provedores não-OpenAI; `sistema` → `usuário` para GLM/ERNIE --**Think Tag Extraction**— Extrai blocos `` de modelos como DeepSeek R1 em `reasoning_content` padronizado --**Saída estruturada para Gemini**— `json_schema` → conversão automática `responseMimeType`/`responseSchema` --**`stream` tem como padrão `false`**— Alinha-se com a especificação OpenAI, evitando SSE inesperado em SDKs Python/Rust/Go
+**How OmniRoute solves it:** - -🌐 3. "Meu provedor de IA bloqueia minha região/país" +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Provedores como OpenAI/Codex bloqueiam o acesso de determinadas regiões geográficas. Os usuários recebem erros como `unsupported_country_region_territory` durante conexões OAuth e API. Isto é especialmente frustrante para desenvolvedores de países em desenvolvimento. + -**Como o OmniRoute resolve isso:** +
+🌐 3. "My AI provider blocks my region/country" --**Configuração de proxy de 3 níveis**— Proxy configurável em 3 níveis: global (todo o tráfego), por provedor (apenas um provedor) e por conexão/chave --**Selos de proxy codificados por cores**— Indicadores visuais: 🟢 proxy global, 🟡 proxy do provedor, 🔵 proxy de conexão, sempre mostrando o IP --**Troca de token OAuth por meio de proxy**— O fluxo OAuth também passa pelo proxy, resolvendo `unsupported_country_region_territory` --**Testes de conexão via proxy**— Os testes de conexão usam o proxy configurado (não há mais bypass direto) --**Suporte SOCKS5**— Suporte completo ao proxy SOCKS5 para roteamento de saída --**TLS Fingerprint Spoofing**— Impressão digital TLS semelhante a um navegador via `wreq-js` para ignorar a detecção de bot --**🔏 CLI Fingerprint Matching**— Reordena cabeçalhos e campos de corpo para corresponder às assinaturas binárias CLI nativas, reduzindo drasticamente o risco de sinalização de conta. O IP do proxy é preservado – você obtém mascaramento de IP furtivo**e**simultaneamente
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. - -🆓 4. "Quero usar IA para codificação, mas não tenho dinheiro" +**How OmniRoute solves it:** -Nem todos podem pagar US$ 20–200/mês por assinaturas de IA. Estudantes, desenvolvedores de países emergentes, amadores e freelancers precisam de acesso a modelos de qualidade a custo zero. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Como o OmniRoute resolve isso:** + --**Provedores de nível gratuito integrados**— Suporte nativo para provedores 100% gratuitos: Qoder (5 modelos ilimitados via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 modelos ilimitados: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID gratuitamente), Gemini CLI (180 mil tokens/mês grátis) --**Ollama Cloud**— Modelos Ollama hospedados na nuvem em `api.ollama.com` com nível gratuito de "uso leve"; use o prefixo `ollamacloud/` --**Combos somente gratuitos**— Cadeia `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/mês com tempo de inatividade zero --**NVIDIA NIM Free Access**— ~40 RPM de acesso gratuito para desenvolvedores para sempre a mais de 70 modelos em build.nvidia.com (transição de créditos para limites de taxa pura) --**Estratégia de Custo Otimizado**— Estratégia de roteamento que escolhe automaticamente o provedor mais barato disponível +
+🆓 4. "I want to use AI for coding but I have no money" - -🔒 5. "Preciso proteger meu gateway de IA contra acesso não autorizado" +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -Ao expor um gateway de IA à rede (LAN, VPS, Docker), qualquer pessoa com o endereço pode consumir os tokens/cota do desenvolvedor. Sem proteção, as APIs ficam vulneráveis ​​ao uso indevido, injeção imediata e abuso. +**How OmniRoute solves it:** -**Como o OmniRoute resolve isso:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**Gerenciamento de chaves de API**— Geração, rotação e escopo por provedor com uma página `/dashboard/api-manager` dedicada --**Permissões em nível de modelo**— Restringir chaves de API a modelos específicos (`openai/*`, padrões curinga), com alternância Permitir tudo/Restringir --**API Endpoint Protection**— Exige uma chave para `/v1/models` e bloqueia provedores específicos da listagem --**Auth Guard + Proteção CSRF**— Todas as rotas do painel protegidas com middleware `withAuth` + tokens CSRF --**Rate Limiter**— Limitação de taxa por IP com janelas configuráveis --**Filtragem de IP**— Lista de permissões/lista de bloqueio para controle de acesso --**Prompt Injection Guard**— Sanitização contra padrões de prompt maliciosos --**Criptografia AES-256-GCM**— Credenciais criptografadas em repouso
+ - -🛑 6. "Meu provedor caiu e perdi meu fluxo de codificação" +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -Os provedores de IA podem ficar instáveis, retornar erros 5xx ou atingir limites de taxa temporários. Se um desenvolvedor depender de um único provedor, ele será interrompido. Sem disjuntores, tentativas repetidas podem travar o aplicativo. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Como o OmniRoute resolve isso:** +**How OmniRoute solves it:** --**Disjuntor por modelo**— Abertura/fechamento automático com limites configuráveis e resfriamento (Fechado/Aberto/Meio-aberto), com escopo definido por modelo para evitar bloqueios em cascata --**Retirada exponencial**— Atrasos progressivos em novas tentativas --**Rebanho Anti-Trovão**— Proteção Mutex + semáforo contra tempestades de novas tentativas simultâneas --**Combo Fallback Chains**— Se o provedor primário falhar, ele cairá automaticamente na cadeia sem intervenção --**Combo Circuit Breaker**— Desativa automaticamente provedores com falha em uma cadeia de combinação --**Health Dashboard**— Monitoramento de tempo de atividade, estados de disjuntores, bloqueios, estatísticas de cache, latência p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest - -🔧 7. "Configurar cada ferramenta de IA é tedioso e repetitivo" + -Os desenvolvedores usam Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Cada ferramenta precisa de uma configuração diferente (endpoint da API, chave, modelo). Reconfigurar ao trocar de provedor ou modelo é uma perda de tempo. +
+🛑 6. "My provider went down and I lost my coding flow" -**Como o OmniRoute resolve isso:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**CLI Tools Dashboard**— Página dedicada com configuração de um clique para Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline --**GitHub Copilot Config Generator**— Gera `chatLanguageModels.json` para VS Code com seleção de modelo em massa --**Assistente de integração**— Configuração guiada em 4 etapas para usuários iniciantes --**Um endpoint, todos os modelos**— Configure `http://localhost:20128/v1` uma vez, acesse mais de 60 provedores
+**How OmniRoute solves it:** - -🔑 8. "Gerenciar tokens OAuth de vários provedores é um inferno" +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot — todos usam OAuth 2.0 com tokens expirados. Os desenvolvedores precisam se autenticar novamente constantemente, lidar com `client_secret is missing`, `redirect_uri_mismatch` e falhas em servidores remotos. OAuth em LAN/VPS é particularmente problemático. + -**Como o OmniRoute resolve isso:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Atualização automática de token**— Os tokens OAuth são atualizados em segundo plano antes da expiração --**OAuth 2.0 (PKCE) integrado**— Fluxo automático para Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder --**OAuth de várias contas**— Várias contas por provedor por meio de extração de token JWT/ID --**OAuth LAN/Remote Fix**— Detecção de IP privado para `redirect_uri` + modo de URL manual para servidores remotos --**OAuth por trás do Nginx**— Usa `window.location.origin` para compatibilidade de proxy reverso --**Guia OAuth remoto**— Guia passo a passo para credenciais do Google Cloud em VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. - -📊 9. "Não sei quanto estou gastando ou onde" +**How OmniRoute solves it:** -Os desenvolvedores usam vários provedores pagos, mas não têm uma visão unificada dos gastos. Cada provedor possui seu próprio painel de faturamento, mas não há visão consolidada. Custos inesperados podem se acumular. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Como o OmniRoute resolve isso:** + --**Painel de análise de custos**— Acompanhamento de custos por token e gerenciamento de orçamento por provedor --**Limites de orçamento por nível**— Teto de gastos por nível que aciona substituto automático --**Configuração de preços por modelo**— Preços configuráveis por modelo --**Estatísticas de uso por chave de API**— Contagem de solicitações e carimbo de data/hora do último uso por chave --**Painel de análise**— Cartões de estatísticas, gráfico de uso do modelo, tabela de provedores com taxas de sucesso e latência +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" - -🐛 10. "Não consigo diagnosticar erros e problemas em chamadas de IA" +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Quando uma chamada falha, o desenvolvedor não sabe se foi um limite de taxa, um token expirado, um formato errado ou um erro do provedor. Logs fragmentados em diferentes terminais. Sem observabilidade, a depuração é uma tentativa e erro. +**How OmniRoute solves it:** -**Como o OmniRoute resolve isso:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Painel de registros unificados**— 4 guias: registros de solicitação, registros de proxy, registros de auditoria, console --**Console Log Viewer**— Visualizador em estilo terminal em tempo real com níveis codificados por cores, rolagem automática, pesquisa, filtro --**SQLite Proxy Logs**— Logs persistentes que sobrevivem às reinicializações do servidor --**Translator Playground**— 4 modos de depuração: Playground (tradução de formato), Chat Tester (ida e volta), Test Bench (lote), Live Monitor (tempo real) --**Solicitar telemetria**— latência p50/p95/p99 + rastreamento X-Request-Id --**Registro baseado em arquivo com rotação**— Os logs do aplicativo são alternados por tamanho, dias de retenção e contagem de arquivos; os artefatos do registro de chamadas são alternados por dias de retenção e contagem de arquivos --**Relatório de informações do sistema**— `npm run system-info` gera `system-info.txt` com seu ambiente completo (versão do Node, versão do OmniRoute, sistema operacional, ferramentas CLI, status do Docker/PM2). Anexe-o ao relatar problemas para triagem instantânea.
+ - -🏗️ 11. "Implantar e manter o gateway é complexo" +
+📊 9. "I don't know how much I'm spending or where" -Instalar, configurar e manter um proxy de IA em diferentes ambientes (local, VPS, Docker, nuvem) exige muito trabalho. Problemas como caminhos codificados, `EACCES` em diretórios, conflitos de porta e compilações entre plataformas adicionam atrito. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Como o OmniRoute resolve isso:** +**How OmniRoute solves it:** --**instalação global npm**— `npm install -g omniroute && omniroute` — concluído --**Docker Multiplataforma**— AMD64 + ARM64 nativo (Apple Silicon, AWS Graviton, Raspberry Pi) --**Docker Compose Profiles**— `base` (sem ferramentas CLI) e `cli` (com Claude Code, Codex, OpenClaw) --**Aplicativo Electron Desktop**— Aplicativo nativo para Windows/macOS/Linux com bandeja do sistema, inicialização automática e modo offline --**Modo Split-Port**— API e Dashboard em portas separadas para cenários avançados (proxy reverso, rede de contêineres) --**Cloud Sync**— Sincronização de configuração entre dispositivos via Cloudflare Workers --**Backups de banco de dados**— Backup, restauração, exportação e importação automática de todas as configurações, com `DISABLE_SQLITE_AUTO_BACKUP` para backups gerenciados externamente
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency - -🌍 12. "A interface é somente em inglês e minha equipe não fala inglês" + -Equipes em países que não falam inglês, especialmente na América Latina, Ásia e Europa, enfrentam dificuldades com interfaces somente em inglês. As barreiras linguísticas reduzem a adoção e aumentam os erros de configuração. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Como o OmniRoute resolve isso:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Painel i18n — 30 idiomas**— Todas as mais de 500 teclas traduzidas, incluindo árabe, búlgaro, dinamarquês, alemão, espanhol, finlandês, francês, hebraico, hindi, húngaro, indonésio, italiano, japonês, coreano, malaio, holandês, norueguês, polonês, português (PT/BR), romeno, russo, eslovaco, sueco, tailandês, ucraniano, vietnamita, chinês, filipino, inglês --**Suporte RTL**— Suporte da direita para a esquerda para árabe e hebraico --**READMEs multilíngues**— 30 traduções completas de documentação --**Seletor de idioma**— Ícone de globo no cabeçalho para troca em tempo real
+**How OmniRoute solves it:** - -🔄 13. "Preciso de mais do que bate-papo - preciso de incorporações, imagens, áudio" +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -IA não é apenas conclusão de bate-papo. Os desenvolvedores precisam gerar imagens, transcrever áudio, criar embeddings para RAG, reclassificar documentos e moderar conteúdo. Cada API possui um endpoint e formato diferente. + -**Como o OmniRoute resolve isso:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Embeddings**— `/v1/embeddings` com 6 provedores e mais de 9 modelos --**Geração de imagens**— `/v1/images/Generations` com 10 provedores e mais de 20 modelos (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Texto para vídeo**— `/v1/videos/generações` — ComfyUI (AnimateDiff, SVD) e SD WebUI --**Text-to-Music**— `/v1/music/Generations` — ComfyUI (Stable Audio Open, MusicGen) --**Transcrição de áudio**— `/v1/audio/transcrições` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Text-to-Speech**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + provedores existentes --**Moderações**— `/v1/moderations` — Verificações de segurança de conteúdo --**Reclassificação**— `/v1/rerank` — Reclassificação da relevância do documento --**API Responses**— Suporte completo `/v1/responses` para Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. - -🧪 14. "Não tenho como testar e comparar a qualidade entre modelos" +**How OmniRoute solves it:** -Os desenvolvedores querem saber qual modelo é melhor para seu caso de uso – código, tradução, raciocínio – mas comparar manualmente é lento. Não existem ferramentas de avaliação integradas. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**Como o OmniRoute resolve isso:** + --**Avaliações LLM**— Teste Golden Set com 10 casos pré-carregados cobrindo saudações, matemática, geografia, geração de código, conformidade com JSON, tradução, remarcação, recusa de segurança --**4 estratégias de correspondência**— `exato`, `contém`, `regex`, `custom` (função JS) --**Translator Playground Test Bench**— Teste em lote com múltiplas entradas e saídas esperadas, comparação entre fornecedores --**Testador de bate-papo**— Ida e volta completa com renderização de resposta visual --**Monitoramento ao vivo**— Transmissão em tempo real de todas as solicitações que passam pelo proxy +
+🌍 12. "The interface is English-only and my team doesn't speak English" - -📈 15. "Preciso escalar sem perder desempenho" +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -À medida que o volume de solicitações aumenta, sem armazenar em cache as mesmas perguntas geram custos duplicados. Sem idempotência, solicitações duplicadas desperdiçam processamento. Os limites de tarifas por provedor devem ser respeitados. +**How OmniRoute solves it:** -**Como o OmniRoute resolve isso:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Cache Semântico**— Cache de duas camadas (assinatura + semântica) reduz custo e latência --**Idempotência de solicitação**— janela de desduplicação de 5s para solicitações idênticas --**Detecção de limite de taxa**— RPM por provedor, intervalo mínimo e rastreamento simultâneo máximo --**Limites de taxa editáveis**— Padrões configuráveis em Configurações → Resiliência com persistência --**Cache de validação de chave de API**— cache de três camadas para desempenho de produção --**Health Dashboard com telemetria**— latência p50/p95/p99, estatísticas de cache, tempo de atividade
+ - -🤖 16. "Quero controlar o comportamento do modelo globalmente" +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Desenvolvedores que desejam todas as respostas em um idioma específico, com um tom específico ou que desejam limitar os tokens de raciocínio. Configurar isso em cada ferramenta/solicitação é impraticável. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Como o OmniRoute resolve isso:** +**How OmniRoute solves it:** --**Injeção de Prompt do Sistema**— Prompt global aplicado a todas as solicitações --**Thinking Budget Validation**— Controle de alocação de token de raciocínio por solicitação (passthrough, automático, personalizado, adaptativo) --**9 Estratégias de Roteamento**— Estratégias globais que determinam como as solicitações são distribuídas --**Wildcard Router**— Os padrões `provider/*` roteiam dinamicamente para qualquer provedor --**Combo Habilitar/Desabilitar Alternar**— Alternar combos diretamente do painel --**Alternância de provedor**— Habilite/desabilite todas as conexões de um provedor com um clique --**Provedores bloqueados**— Excluir provedores específicos da listagem `/v1/models`
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex - -🧰 17. "Preciso de ferramentas MCP como recursos de produto de primeira classe" + -Muitos gateways de IA expõem o MCP apenas como um detalhe de implementação oculto. As equipes precisam de uma camada operacional visível e gerenciável. +
+🧪 14. "I have no way to test and compare quality across models" -**Como o OmniRoute resolve isso:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP aparece na navegação do painel e na guia protocolo de endpoint -- Página dedicada de gerenciamento de MCP com processos, ferramentas, escopos e auditoria -- Início rápido integrado para `omniroute --mcp` e integração do cliente
+**How OmniRoute solves it:** - -🧠 18. "Preciso de orquestração A2A com caminhos de tarefa de sincronização + fluxo" +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Os fluxos de trabalho do agente precisam de respostas diretas e execução em streaming de longa duração com controle do ciclo de vida. + -**Como o OmniRoute resolve isso:** +
+📈 15. "I need to scale without losing performance" -- Endpoint A2A JSON-RPC (`POST /a2a`) com `message/send` e `message/stream` -- Streaming SSE com propagação de estado terminal -- APIs de ciclo de vida de tarefas para `tasks/get` e `tasks/cancel`
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. - -🛰️ 19. "Preciso de integridade real do processo MCP, não de status adivinhado" +**How OmniRoute solves it:** -As equipes operacionais precisam saber se o MCP está realmente ativo, e não apenas se uma API está acessível. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**Como o OmniRoute resolve isso:** + -- Arquivo de pulsação em tempo de execução com PID, carimbos de data/hora, transporte, contagem de ferramentas e modo de escopo -- API de status MCP combinando pulsação + atividade recente -- Cartões de status da interface do usuário para atualização de processo/tempo de atividade/pulsação +
+🤖 16. "I want to control model behavior globally" - -📋 20. "Preciso de execução auditável da ferramenta MCP" +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Quando as ferramentas alteram a configuração ou acionam ações operacionais, as equipes precisam de rastreabilidade forense. +**How OmniRoute solves it:** -**Como o OmniRoute resolve isso:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- Registro de auditoria apoiado por SQLite para chamadas de ferramentas MCP -- Filtros por ferramenta, sucesso/falha, chave de API e paginação -- Tabela de auditoria do painel + endpoints de estatísticas para automação
+ - -🔐 21. "Preciso de permissões MCP com escopo definido por integração" +
+🧰 17. "I need MCP tools as first-class product capabilities" -Clientes diferentes devem ter acesso com privilégios mínimos às categorias de ferramentas. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Como o OmniRoute resolve isso:** +**How OmniRoute solves it:** -- 10 escopos MCP granulares para acesso controlado à ferramenta -- Aplicação do escopo e visibilidade na UI de gerenciamento do MCP -- Postura padrão segura para ferramentas operacionais
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding - -⚙️ 22. "Preciso de controles operacionais sem reimplantar" + -As equipes precisam de mudanças rápidas no tempo de execução durante incidentes ou eventos de custo. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**Como o OmniRoute resolve isso:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Alternar ativação combinada diretamente do painel MCP -- Aplicar perfis de resiliência de pacotes de políticas predefinidos -- Redefinir o estado do disjuntor no mesmo painel de operações
+**How OmniRoute solves it:** - -🔄 23. "Preciso de visibilidade e cancelamento do ciclo de vida da tarefa A2A ao vivo" +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Sem visibilidade do ciclo de vida, os incidentes de tarefas tornam-se difíceis de triagem. + -**Como o OmniRoute resolve isso:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Listagem/filtragem de tarefas por estado/habilidade com paginação -- Detalhamento de metadados de tarefas, eventos e artefatos -- Terminal de cancelamento de tarefa e ação de UI com confirmação
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. - -🌊 24. "Preciso de métricas de fluxo ativo para carga A2A" +**How OmniRoute solves it:** -Os fluxos de trabalho de streaming exigem insights operacionais sobre simultaneidade e conexões em tempo real. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**Como o OmniRoute resolve isso:** + -- Contadores de fluxo ativos integrados ao status A2A -- Carimbo de data/hora da última tarefa e contagens por estado -- Cartões de painel A2A para monitoramento de operações em tempo real +
+📋 20. "I need auditable MCP tool execution" - -🪪 25. "Preciso de descoberta de agente padrão para clientes" +When tools mutate config or trigger ops actions, teams need forensic traceability. -Clientes e orquestradores externos precisam de metadados legíveis por máquina para integração. +**How OmniRoute solves it:** -**Como o OmniRoute resolve isso:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- Cartão de agente exposto em `/.well-known/agent.json` -- Capacidades e habilidades mostradas na UI de gerenciamento -- A API de status A2A inclui metadados de descoberta para automação
+ - -🧭 26. "Preciso de descoberta de protocolo na experiência do usuário do produto" +
+🔐 21. "I need scoped MCP permissions per integration" -Se os usuários não conseguirem descobrir superfícies de protocolo, a adoção e a qualidade do suporte cairão. +Different clients should have least-privilege access to tool categories. -**Como o OmniRoute resolve isso:** +**How OmniRoute solves it:** -- Página**Endpoints**consolidada com guias para Proxy, MCP, A2A e API Endpoints -- Alterna o status do serviço inline (Online/Offline) para MCP e A2A -- Links da visão geral para guias de gerenciamento dedicadas
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling - -🧪 27. "Preciso de validação de protocolo ponta a ponta com clientes reais" + -Os testes simulados não são suficientes para validar a compatibilidade do protocolo antes do lançamento. +
+⚙️ 22. "I need operational controls without redeploying" -**Como o OmniRoute resolve isso:** +Teams need quick runtime changes during incidents or cost events. -- Suíte E2E que inicializa o aplicativo e usa transporte de cliente SDK MCP real -- Testes de cliente A2A para fluxos de descoberta, envio, streaming, obtenção e cancelamento -- Verificação cruzada de afirmações com APIs de auditoria MCP e tarefas A2A
+**How OmniRoute solves it:** - -📡 28. "Preciso de observabilidade unificada em todas as interfaces" +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -A divisão da observabilidade por protocolo cria pontos cegos e MTTR mais longo. + -**Como o OmniRoute resolve isso:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Painéis/logs/análises unificados em um produto -- Saúde + auditoria + solicitação de telemetria nas camadas OpenAI, MCP e A2A -- APIs operacionais para status e automação
+Without lifecycle visibility, task incidents become hard to triage. - -💼 29. "Preciso de um tempo de execução para proxy + ferramentas + orquestração de agentes" +**How OmniRoute solves it:** -A execução de muitos serviços separados aumenta o custo operacional e os modos de falha. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**Como o OmniRoute resolve isso:** + -- Proxy compatível com OpenAI, servidor MCP e servidor A2A em uma pilha -- Autenticação compartilhada, resiliência, armazenamento de dados e observabilidade -- Modelo de política consistente em todas as superfícies de interação +
+🌊 24. "I need active stream metrics for A2A load" - -🚀 30. "Preciso enviar fluxos de trabalho de agente sem confusão de códigos colados" +Streaming workflows require operational insight into concurrency and live connections. -As equipes perdem velocidade ao unir vários serviços e scripts ad-hoc. +**How OmniRoute solves it:** -**Como o OmniRoute resolve isso:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Estratégia unificada de endpoint para clientes e agentes -- UIs de gerenciamento de protocolo integradas e caminhos de validação de fumaça -- Fundações prontas para produção (segurança, registro, resiliência, backup)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Manual A: Maximize a assinatura paga + backup barato**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -607,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Manual B: Pilha de codificação de custo zero**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Manual C: cadeia de fallback sempre ativa 24 horas por dia, 7 dias por semana**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -630,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Manual D: Operações de agente com MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Configure a codificação de IA em minutos por**$0/mês**. Conecte essas contas gratuitas e use o combo**Free Stack**integrado. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Etapa | Ação | Provedores desbloqueados | +| Step | Action | Providers Unlocked | | ---- | -------------------------------------------------- | ------------------------------------------------------------------ | -| 1 | Conectar**Kiro**(ID do AWS Builder OAuth) | Claude Soneto 4.5, Haiku 4.5 —**ilimitado**| -| 2 | Conecte**Qoder**(Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... —**ilimitado**| -| 3 | Conecte**Qwen**(código do dispositivo) | qwen3-coder-plus, qwen3-coder-flash... —**ilimitado**| -| 4 | Conecte**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180K/mês grátis**| -| 5 | `/dashboard/combos` →**Pilha grátis ($0)**modelo | Round-robin todos os provedores gratuitos automaticamente | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Aponte qualquer IDE/CLI para:**`http://localhost:20128/v1` · Chave API: `any-string` · Concluído. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Cobertura extra opcional (também gratuita):**Chave de API Groq (30 RPM grátis), NVIDIA NIM (40 RPM grátis, modelos com mais de 70), Cerebras (1 milhão de tok/dia), chave de API LongCat (50 milhões de tokens/dia!), Cloudflare Workers AI (10 mil neurônios/dia, mais de 50 modelos).## Início Rápido +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Início Rápido ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **usuários pnpm:**Execute `pnpm aprovar-builds -g` após a instalação para habilitar scripts de construção nativos exigidos por `better-sqlite3` e `@swc/core`: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > > ```bash -> instalação pnpm -g omniroute -> pnpm aprovar-builds -g # Selecione todos os pacotes → aprovar -> omnirota +> pnpm install -g omniroute +> pnpm approve-builds -g # Select all packages → approve +> omniroute > ``` -O painel abre em `http://localhost:20128` e o URL base da API é `http://localhost:20128/v1`. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Comando | Descrição | -| ----------------------- | --------------------------------------------------------------- | -| `omnirota` | Iniciar servidor (`PORT=20128`, API e dashboard na mesma porta) | -| `omniroute --port 3000` | Defina a porta canônica/API como 3000 | -| `omniroute --mcp` | Inicie o servidor MCP (transporte stdio) | -| `omniroute --no-open` | Não abra o navegador automaticamente | -| `omniroute --help` | Mostrar ajuda | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Modo de porta dividida opcional:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -Para a maioria das implantações, você só precisa de: +For most deployments, you only need: -| Variável | Padrão | Finalidade | -| ------------------------ | ----------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600000` | Linha de base compartilhada para busca upstream, tempos limite ocultos de Undici, solicitações de impressão digital TLS e tempos limite de solicitação/proxy de ponte de API | -| `STREAM_IDLE_TIMEOUT_MS` | herda `REQUEST_TIMEOUT_MS` | Intervalo máximo entre blocos de streaming antes que o OmniRoute anule o fluxo SSE | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -A compatibilidade com versões anteriores é preservada: `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` e outras variáveis ​​de tempo limite por camada ainda funcionam e substituem a linha de base compartilhada. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Substituições avançadas estão disponíveis se você precisar de um controle mais preciso:| Variável | Padrão | Finalidade | +Advanced overrides are available if you need finer control: + +| Variable | Default | Purpose | | ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | herda `REQUEST_TIMEOUT_MS` | Tempo limite total da solicitação upstream usado pelo sinal de aborto de busca principal | -| `FETCH_HEADERS_TIMEOUT_MS` | herda `FETCH_TIMEOUT_MS` | Prazo Undici para recepção de cabeçalhos de resposta a montante | -| `FETCH_BODY_TIMEOUT_MS` | herda `FETCH_TIMEOUT_MS` | Limite de tempo Undici entre partes do corpo upstream (`0` desativa) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Tempo limite de conexão TCP Undici | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici tempo limite de soquete keep-alive ocioso | -| `TLS_CLIENT_TIMEOUT_MS` | herda `FETCH_TIMEOUT_MS` | Tempo limite para solicitações de impressão digital TLS feitas por meio de `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | herda `REQUEST_TIMEOUT_MS` ou `30000` | Tempo limite para encaminhamento de proxy `/v1` da porta API para a porta do painel | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `máx(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Tempo limite de solicitação de entrada no servidor ponte API | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Tempo limite do cabeçalho de entrada no servidor ponte API | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Tempo limite de atividade no servidor ponte API | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Tempo limite de inatividade do soquete no servidor ponte API (`0` desativa) | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -Se você executar o OmniRoute atrás de Nginx, Caddy, Cloudflare ou outro proxy reverso, certifique-se de que o proxy -os tempos limite também são maiores que os tempos limites de fluxo/busca do OmniRoute.### 2) Connect providers and create your API key +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. -1. Abra Dashboard → `Providers` e conecte pelo menos um provedor (OAuth ou chave API). -2. Abra Dashboard → `Endpoints` e crie uma chave de API. -3. (Opcional) Abra Dashboard → `Combos` e defina sua cadeia de fallback.### 3) Point your coding tool to OmniRoute +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Funciona com Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode e SDKs compatíveis com OpenAI.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (para operações orientadas por ferramentas):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -Em seguida, conecte seu cliente MCP através de `stdio` e teste ferramentas como: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (para fluxos de trabalho entre agentes):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -759,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Este conjunto valida fluxos reais de clientes MCP e A2A em um aplicativo em execução.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -767,13 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` - -Void Linux (modelo `xbps-src`) +
+Void Linux (`xbps-src` template) -Para usuários do Void Linux, você pode construir um pacote nativo usando `xbps-src`. Salve este bloco como `srcpkgs/omniroute/template`:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -785,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -793,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -869,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -880,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute está disponível como uma imagem pública do Docker no [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Execução rápida:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -890,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**Com arquivo de ambiente:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Usando Docker Compose:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -O suporte do painel para implantações do Docker agora inclui um**Cloudflare Quick Tunnel**com um clique em `Dashboard → Endpoints`. A primeira habilita o download de `cloudflared` somente quando necessário, inicia um túnel temporário para seu endpoint `/v1` atual e mostra o URL `https://*.trycloudflare.com/v1` gerado diretamente abaixo de seu URL público normal. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Notas: +Notes: -- Os URLs do Quick Tunnel são temporários e mudam após cada reinicialização. -- Os Quick Tunnels não são restaurados automaticamente após um OmniRoute ou reinicialização do contêiner. Reative-os no painel quando necessário. -- A instalação gerenciada atualmente oferece suporte a Linux, macOS e Windows em `x64` / `arm64`. -- Túneis rápidos gerenciados padrão para transporte HTTP/2 para evitar avisos de buffer QUIC UDP barulhentos em ambientes de contêiner restritos. Defina `CLOUDFLARED_PROTOCOL=quic` ou `auto` se desejar um transporte diferente. -- As imagens do Docker agrupam as raízes da CA do sistema e as passam para o `cloudflared` gerenciado, o que evita falhas de confiança do TLS quando o túnel é inicializado dentro do contêiner. -- SQLite roda em modo WAL. `docker stop` deve ter permissão para terminar para que o OmniRoute possa verificar as alterações mais recentes em `storage.sqlite`. -- Os arquivos Compose incluídos já definem um período de carência de 40 segundos. Se você executar a imagem diretamente, mantenha `--stop-timeout 40` (ou similar) para que as paradas manuais não interrompam a limpeza do desligamento. -- Defina `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` se quiser que o OmniRoute use um binário existente em vez de baixar um. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Usando Docker Compose com Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute pode ser exposto com segurança usando o provisionamento SSL automático da Caddy. Certifique-se de que o registro DNS A do seu domínio aponte para o IP do seu servidor.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` - -| Imagem | Etiqueta | Tamanho | Descrição | +| Image | Tag | Size | Description | | ------------------------ | -------- | ------ | --------------------- | -| `diegosouzapw/omniroute` | `mais recente` | ~250 MB | Última versão estável | -| `diegosouzapw/omniroute` | `1.0.3` | ~250 MB | Versão atual |--- +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | + +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**NOVO!**OmniRoute agora está disponível como um**aplicativo de desktop nativo**para Windows, macOS e Linux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Execute o OmniRoute como um aplicativo de desktop independente — sem terminal, sem navegador, sem necessidade de internet para modelos locais. O aplicativo baseado em Electron inclui: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Janela Nativa**— Janela de aplicativo dedicada com integração na bandeja do sistema -- 🔄**Início automático**— Inicie o OmniRoute no login do sistema -- 🔔**Notificações nativas**— Receba alertas sobre esgotamento de cota ou problemas com o provedor -- ⚡**Instalação com um clique**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Modo offline**— Funciona totalmente offline com servidor incluído### Início Rápido +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Início Rápido ```bash # Development mode @@ -979,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -Quando minimizado, o OmniRoute fica na bandeja do sistema com ações rápidas: +When minimized, OmniRoute lives in your system tray with quick actions: -- Abra o painel -- Alterar porta do servidor -- Sair do aplicativo +- Open dashboard +- Change server port +- Quit application -📖 Documentação completa: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Nível | Provedor | Custo | Redefinição de cota | Melhor para | -| ------------------- | ------------------------------------- | ------------------------------------- | ------------------------ | ----------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 ASSINATURA** | Código Claude (Pro) | $ 20/mês | 5h + semanalmente | Já inscrito | -| | Códice (Plus/Pro) | US$ 20-200/mês | 5h + semanalmente | Usuários OpenAI | -| | Gêmeos CLI | **GRÁTIS** | 180 mil/mês + 1 mil/dia | Todos! | -| | Copiloto GitHub | US$ 10-19/mês | Mensalmente | Usuários do GitHub | -| **🔑 CHAVE DE API** | NVIDIA NIM | **GRÁTIS**(desenvolvedor para sempre) | ~40RPM | Mais de 70 modelos abertos | -| | Cérebros | **GRÁTIS**(1 milhão de tok/dia) | 60KTPM/30RPM | O mais rápido do mundo | -| | Groq | **GRÁTIS**(30 RPM) | RPD de 14,4K | Lhama/Gemma ultrarrápida | -| | DeepSeek V3.2 | US$ 0,27/US$ 1,10 por 1 milhão | Nenhum | Melhor raciocínio preço/qualidade | -| | xAI Grok-4 Rápido | **$0,20/$0,50 por 1 milhão**🆕 | Nenhum | Chamada de ferramenta mais rápida +, ultrabaixa | -| | xAI Grok-4 (padrão) | US$ 0,20/US$ 1,50 por 1 milhão 🆕 | Nenhum | Carro-chefe do raciocínio da xAI | -| | Mistral | Teste grátis + pago | Taxa limitada | IA Europeia | -| | OpenRouter | Pagamento conforme uso | Nenhum | Mais de 100 modelos no total. | -| **💰 BARATO** | GLM-5 (via Z.AI) 🆕 | US$ 0,5/1 milhão | Diariamente 10h | Saída de 128K, o mais novo carro-chefe | -| | GLM-4.7 | US$ 0,6/1 milhão | Diariamente 10h | Backup de orçamento | -| | MiniMax M2.5 🆕 | Entrada de US$ 0,3/1 milhão | Rolamento de 5 horas | Raciocínio + tarefas de agência | -| | MiniMax M2.1 | US$ 0,2/1 milhão | Rolamento de 5 horas | Opção mais barata | -| | Kimi K2.5 (API Moonshot) 🆕 | Pagamento conforme uso | Nenhum | Acesso direto à API Moonshot | -| | Kimi K2 | $ 9 / mês fixo | 10 milhões de tokens/mês | Custo previsível | -| **🆓 GRÁTIS** | Qoder | **$0** | Ilimitado | 5 modelos ilimitados | -| | Qwen | **$0** | Ilimitado | 4 modelos ilimitados | -| | Kiro | **$0** | Ilimitado | Claude Sonnet/Haiku (Construtor AWS) | -| | LongCat Flash Lite 🆕 | **$0**(50 milhões de dólares/dia 🔥) | 1RPS | Maior cota gratuita do planeta | -| | Polinizações AI 🆕 | **$0**(sem necessidade de chave) | 1 necessidade/15s | GPT-5, Claude, DeepSeek, Lhama 4 | -| | IA dos trabalhadores da Cloudflare 🆕 | **$0**(10 mil neurônios/dia) | ~150 resp/dia | Mais de 50 modelos, vantagem global | -| | IA Scaleway 🆕 | **$0**(total de 1 milhão de tokens) | Taxa limitada | UE/GDPR, Qwen3 235B, Llama 70B | > 🆕**Novos modelos adicionados (março de 2026):**Família Grok-4 Fast a US$ 0,20/US$ 0,50/M (comparado em 1143ms — 30% mais rápido que Gemini 2.5 Flash), GLM-5 via Z.AI com saída de 128K, raciocínio MiniMax M2.5, preço atualizado DeepSeek V3.2, Kimi K2.5 via API direta Moonshot. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 Pilha Combo de $0 — A configuração gratuita completa:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Custo zero. Nunca para de codificar.**Configure isso como um combo OmniRoute e todos os fallbacks acontecem automaticamente - nunca há troca manual.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Todos os modelos abaixo são**100% gratuitos, sem necessidade de cartão de crédito**. OmniRoute roteia automaticamente entre eles quando uma cota acaba – combine todos eles para um combo inquebrável de $ 0.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Modelo | Prefixo | Limite | Limite de taxa | +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | --------------------- | -| `claude-soneto-4.5` | `kr/` |**Ilimitado**| Nenhum limite diário comunicado | -| `claude-haiku-4.5` | `kr/` |**Ilimitado**| Nenhum limite diário comunicado | -| `claude-opus-4.6` | `kr/` |**Ilimitado**| Último Opus via Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -| Modelo | Prefixo | Limite | Limite de taxa | +### 🟢 QODER MODELS (Free PAT via qodercli) + +| Model | Prefix | Limit | Rate Limit | | ------------------ | ------ | ------------- | --------------- | -| `kimi-k2-pensando` | `se/` |**Ilimitado**| Nenhum limite máximo comunicado | -| `qwen3-coder-plus` | `se/` |**Ilimitado**| Nenhum limite máximo comunicado | -| `deepseek-r1` | `se/` |**Ilimitado**| Nenhum limite máximo comunicado | -| `minimax-m2.1` | `se/` |**Ilimitado**| Nenhum limite máximo comunicado | -| `kimi-k2` | `se/` |**Ilimitado**| Nenhum limite máximo comunicado | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -> Método de conexão recomendado:**Token de acesso pessoal + `qodercli`**. OAuth do navegador é -> experimental e desativado por padrão, a menos que variáveis de ambiente `QODER_OAUTH_*` sejam configuradas.### 🟡 QWEN MODELS (Device Code Auth) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| Modelo | Prefixo | Limite | Limite de taxa | +### 🟡 QWEN MODELS (Device Code Auth) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | ------------------- | -| `qwen3-coder-plus` | `qw/` |**Ilimitado**| Nenhum limite máximo comunicado | -| `qwen3-codificador-flash` | `qw/` |**Ilimitado**| Nenhum limite máximo comunicado | -| `qwen3-coder-next` | `qw/` |**Ilimitado**| Nenhum limite máximo comunicado | -| `modelo de visão` | `qw/` |**Ilimitado**| Multimodal (imagens) |### 🟣 GEMINI CLI (Google OAuth) +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Modelo | Prefixo | Limite | Limite de taxa | +### 🟣 GEMINI CLI (Google OAuth) + +| Model | Prefix | Limit | Rate Limit | | ------------------------ | ------ | --------------------------- | ------------- | -| `gemini-3-flash-preview` | `gc/` |**180 mil tok/mês**+ 1 mil/dia | Redefinição mensal | -| `gemini-2.5-pro` | `gc/` | 180 mil/mês (pool compartilhado) | Alta qualidade |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -| Nível | Limite Diário | Limite de taxa | Notas | +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---------- | ------------ | ----------- | ------------------------------------------------------ | -| Grátis (Desenvolvedor) | Sem limite de token |**~40RPM**| Mais de 70 modelos; transição para limites de taxas puras em meados de 2025 | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -Modelos gratuitos populares: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -| Nível | Limite Diário | Limite de taxa | Notas | +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) + +| Tier | Daily Limit | Rate Limit | Notes | | ---- | ----------------- | ---------------- | ------------------------------------------- | -| Grátis |**1 milhão de tokens/dia**| 60KTPM/30RPM | A inferência LLM mais rápida do mundo; reinicia diariamente | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -Disponível gratuitamente: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -| Nível | Limite Diário | Limite de taxa | Notas | +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---- | ------------- | ---------------- | ----------------------------------------- | -| Grátis |**RPD de 14,4K**| 30 RPM por modelo | Sem cartão de crédito; 429 no limite, sem cobrança | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -Disponível gratuitamente: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -| Modelo | Prefixo | Cota diária gratuita | Notas | +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 + +| Model | Prefix | Daily Free Quota | Notes | | ----------------------------- | ------ | ----------------- | ----------------------- | -| `LongCat-Flash-Lite` | `lc/` |**50 milhões de tokens**💥 | Maior cota gratuita de todos os tempos | -| `LongCat-Flash-Chat` | `lc/` | 500 mil tokens | Bate-papo multiturno | -| `LongCat-Flash-Thinking` | `lc/` | 500 mil tokens | Raciocínio / CoT | -| `LongCat-Flash-Thinking-2601` | `lc/` | 500 mil tokens | Versão de janeiro de 2026 | -| `LongCat-Flash-Omni-2603` | `lc/` | 500 mil tokens | Multimodal | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | -> 100% gratuito durante a versão beta pública. Cadastre-se em [longcat.chat](https://longcat.chat) com e-mail ou telefone. Reinicia diariamente às 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. -| Modelo | Prefixo | Limite de taxa | Provedor por trás | +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| `openai` | `pol/` | 1 necessidade/15s | GPT-5 | -| `cláudio` | `pol/` | 1 necessidade/15s | Claude Antrópico | -| `gêmeos` | `pol/` | 1 necessidade/15s | Google Gêmeos | -| `busca profunda` | `pol/` | 1 necessidade/15s | DeepSeek V3 | -| `lhama` | `pol/` | 1 necessidade/15s | Batedor Meta Lhama 4 | -| `mistral` | `pol/` | 1 necessidade/15s | IA Mistral | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**Atrito zero:**Sem inscrição, sem chave de API. Adicione o provedor Polinizações com um campo-chave vazio e ele funcionará imediatamente.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| Nível | Neurônios Diários | Uso equivalente | Notas | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 + +| Tier | Daily Neurons | Equivalent Usage | Notes | | ---- | ------------- | --------------------------------------- | ----------------------- | -| Grátis |**10.000**| ~150 LLM resp / áudio 500s / incorporações de 15K | Vantagem global, mais de 50 modelos | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -Modelos gratuitos populares: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (áudio grátis!), `@cf/qwen/qwen2.5-coder-15b-instruct` +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -> Requer token de API + ID da conta de [dash.cloudflare.com](https://dash.cloudflare.com). Armazene o ID da conta nas configurações do provedor.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -| Nível | Cota Grátis | Localização | Notas | +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | | ---- | ------------- | ------------ | ----------------------------------- | -| Grátis |**1 milhão de tokens**| 🇫🇷 Paris, UE | Não é necessário cartão de crédito dentro dos limites | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | -Disponível gratuitamente: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` -> Compatível com UE/GDPR. Obtenha a chave de API em [console.scaleway.com](https://console.scaleway.com). +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). ->**💡 The Ultimate Free Stack (11 provedores, $ 0 para sempre):** +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Sonnet/Haiku ILIMITADO -> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 ILIMITADO -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 milhões de tokens/dia 🔥 -> Polinizações (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — nenhuma chave necessária -> Qwen (qw /) → modelos de codificador qwen3 ILIMITADOS -> Gemini (gemini/) → Gemini 2.5 Flash – 1.500 req/dia grátis -> Cloudflare AI (cf/) → mais de 50 modelos — 10 mil neurônios/dia -> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1 milhão de tokens grátis (UE) -> Groq (groq/) → Llama/Gemma — 14,4K req/dia ultrarrápido -> NVIDIA NIM (nvidia/) → Mais de 70 modelos abertos — 40 RPM para sempre -> Cerebras (cerebras/) → Llama/Qwen mais rápido do mundo — 1 milhão tok/dia -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Transcreva qualquer áudio/vídeo por**US$ 0**— Deepgram lidera com US$ 200 grátis, AssemblyAI US$ 50 substitutos, Groq Whisper como backup de emergência ilimitado. +## 🎙️ Free Transcription Combo -| Provedor | Créditos Grátis | Melhor Modelo | Limite de taxa | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. + +| Provider | Free Credits | Best Model | Rate Limit | | ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | -| 🟢**Deepgram**|**$200 grátis**(inscrição) | `nova-3` — melhor precisão, mais de 30 idiomas | Sem limite de RPM em créditos gratuitos | -| 🔵**AssemblyAI**|**$50 grátis**(inscrição) | `universal-3-pro` — capítulos, sentimento, PII | Sem limite de RPM em créditos gratuitos | -| 🔴**Groque**|**Grátis para sempre**| `whisper-large-v3` — OpenAI Whisper | 30 RPM (taxa limitada) | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | -**Combo sugerido em `/dashboard/combos`:**``` +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Em seguida, em `/dashboard/media` → guia**Transcrição**: carregue qualquer arquivo de áudio ou vídeo → selecione seu endpoint combinado → obtenha a transcrição em formatos suportados.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 é construído como uma plataforma operacional, não apenas um proxy de retransmissão.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Recurso | O que faz | -| --------------------------------------- | --------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Grok-4 Família Rápida** | Modelos xAI por US$ 0,20/US$ 0,50/M – benchmark de 1143 ms (30% mais rápido que Gemini 2.5 Flash) | -| 🧠**GLM-5 via Z.AI** | Contexto de saída de 128K, US$ 0,5/1 milhão – o mais novo carro-chefe da família GLM | -| 🔮**MiniMax M2.5** | Raciocínio + tarefas de agência por US$ 0,30/1 milhão — atualização significativa do M2.1 | -| 🎯**toolCalling Flag por modelo** | `toolCalling: true/false` por modelo no registro — AutoCombo ignora modelos sem capacidade de ferramenta | -| 🌍**Detecção de intenção multilíngue** | Palavras-chave PT/ZH/ES/AR na pontuação AutoCombo — melhor seleção de modelos para conteúdo diferente do inglês | -| 📊**Recursos baseados em benchmarks** | Latência p95 real de solicitações ao vivo alimenta pontuação combinada – AutoCombo aprende com dados reais | -| 🔁**Solicitar desduplicação** | Janela de desduplicação baseada em hash de conteúdo — segura para vários agentes, evita cobranças duplicadas | -| 🔌**Estratégia de roteador conectável** | Interface `RouterStrategy` extensível — adicione lógica de roteamento personalizada como plug-ins | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Recurso | O que faz | -| ---------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Parque Modelo** | Página do painel para testar qualquer modelo diretamente - seletores de provedor/modelo/endpoint, Monaco Editor, streaming, aborto, tempo | -| 🔏**Correspondência de impressão digital CLI** | Ordenação de cabeçalho/corpo por provedor para corresponder às assinaturas CLI nativas — alterne por provedor em Configurações > Segurança.**Seu IP proxy é preservado** | -| 🤝**Suporte ACP (Protocolo Agente Cliente)** | Descoberta de agente CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw e mais 9), gerador de processo, endpoint `/api/acp/agents` | -| 🤖**Painel de Agentes ACP** | Depurar › Página Agentes — grade de 14 agentes com status de instalação, versão, formulário de agente personalizado para qualquer ferramenta CLI. Os usuários do**OpenCode**recebem um botão "Baixar opencode.json" que gera automaticamente uma configuração pronta para uso com todos os modelos disponíveis. | -| 🔧**Roteamento `apiFormat` de modelo personalizado** | Modelos personalizados com `apiFormat: "responses"` agora roteiam corretamente para o tradutor da API Responses | -| 🏢**Isolamento do espaço de trabalho do Codex** | Vários espaços de trabalho do Codex por e-mail — OAuth separa corretamente as conexões por ID do espaço de trabalho | -| 🔄**Atualização automática eletrônica** | O aplicativo de desktop verifica atualizações + instalação automática ao reiniciar | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Recurso | O que faz | -| ----------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**Servidor MCP (25 ferramentas)** | Ferramentas IDE/agente por meio de 3 transportes: stdio, SSE (`/api/mcp/sse`), HTTP Streamable (`/api/mcp/stream`). 18 núcleos + 3 memórias + 4 ferramentas de habilidade | -| 🤝**Servidor A2A (JSON-RPC + SSE)** | Execução de tarefas entre agentes com fluxos de sincronização e streaming | -| 🧭**Página de endpoints consolidados** | Página de gerenciamento com guias com guias Endpoint Proxy, MCP, A2A e API Endpoints | -| 🎚️**Alternativas de ativação/desativação de serviço** | Chaves ON/OFF para MCP e A2A com persistência de configurações (padrão: OFF) | -| 🛰️**Pulsação de tempo de execução do MCP** | Status real do processo (pid, tempo de atividade, idade da pulsação, transporte, modo de escopo) | -| 📋**Trilha de auditoria MCP** | Logs de auditoria filtráveis ​​com sucesso/falha e atribuição de chave | -| 🔐**Aplicação do escopo do MCP** | 10 permissões de escopo granular para acesso controlado a ferramentas | -| 📡**Gerenciamento do ciclo de vida de tarefas A2A** | Listar/filtrar tarefas, inspecionar eventos/artefatos, cancelar tarefas em execução | -| 📋**Descoberta de cartão de agente** | `/.well-known/agent.json` para descoberta automática de clientes | -| 🧪**Arnês de teste do protocolo E2E** | Fluxos reais do cliente MCP SDK + A2A em `test:protocols:e2e` | -| ⚙️**Controles operacionais** | Combinação de interruptores, aplicação de perfis de resiliência, reinicialização de disjuntores a partir de uma superfície de controle | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Recurso | O que faz | -| ---------------------------------------------------------- | ---------------------------------------------------------------------------------------------- | ----------------------- | -| 🎯**Fullback inteligente de 4 camadas** | Roteamento automático: Assinatura → Chave de API → Barato → Grátis | -| 📊**Acompanhamento de cotas em tempo real** | Contagem de tokens ativos + contagem regressiva redefinida por provedor | -| 🔄**Tradução de formato** | OpenAI ↔ Claude ↔ Gemini ↔ Respostas com conversões seguras de esquema | -| 👥**Suporte para múltiplas contas** | Múltiplas contas por provedor com seleção inteligente | -| 🔄**Atualização automática de token** | Os tokens OAuth são atualizados automaticamente com nova tentativa | -| 🎨**Combos Personalizados** | 9 estratégias de balanceamento + controle da cadeia de fallback | -| 🌐**Roteador curinga** | `provedor/*` roteamento dinâmico | -| 🧠**Pensando em controles de orçamento** | Limites de raciocínio de passagem, automático, personalizado e adaptativo | -| 🔀**Alases de modelo** | Aliasing de modelo integrado + personalizado e segurança de migração | -| ⚡**Degradação de fundo** | Encaminhar tarefas em segundo plano de baixa prioridade para modelos mais baratos | -| 🧪**Roteamento inteligente com reconhecimento de tarefas** | Seleção automática de modelo por tipo de conteúdo (codificação/visão/análise/resumo) | -| 🔄**Fluxos de trabalho do agente A2A** | Orquestrador FSM determinístico para execuções de agentes em várias etapas com estado | -| 🔀**Roteamento Adaptativo** | Substituição de estratégia dinâmica com base no volume de tokens e complexidade imediata | -| 🎲**Diversidade de Provedores** | Pontuação de entropia de Shannon equilibrando distribuição de tráfego de combinação automática | -| 💬**Injeção imediata do sistema** | Controles de comportamento globais aplicados de forma consistente | -| 📄**Compatibilidade da API de respostas** | Suporte completo `/v1/responses` para Codex e fluxos de trabalho de agentes avançados | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Recurso | O que faz | -| -------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**Geração de imagens** | `/v1/images/generações` com nuvem e backends locais | -| 📐**Incorporações** | `/v1/embeddings` para pipelines de pesquisa e RAG | -| 🎤**Transcrição de áudio** | `/v1/audio/transcrições` — 7 provedores (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), detecção automática de idioma, suporte a MP4/MP3/WAV | -| 🔊**Conversão de texto em fala** | `/v1/audio/speech` — 10 provedores (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) com mensagens de erro corretas | -| 🎬**Geração de Vídeo** | `/v1/videos/generações` (fluxos de trabalho ComfyUI + SD WebUI) | -| 🎵**Geração Musical** | `/v1/music/Generations` (fluxos de trabalho ComfyUI) | -| 🛡️**Moderações** | verificações de segurança `/v1/moderations` | -| 🔀**Reclassificação** | `/v1/rerank` para pontuação de relevância | -| 🔍**Pesquisa na Web**🆕 | `/v1/search` — 5 provedores (Serper, Brave, Perplexity, Exa, Tavily), mais de 6.500 grátis/mês, failover automático, cache | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Recurso | O que faz | -| ---------------------------------------------- | --------------------------------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**Disjuntores** | Acionamento/recuperação por modelo com controles de limite | -| 🎯**Modelos com reconhecimento de endpoint** | Modelos personalizados declaram endpoints suportados + formato API | -| 🛡️**Rebanho Anti-Trovão** | Proteções Mutex + semáforo em eventos de nova tentativa/taxa | -| 🧠**Semântica + Cache de Assinatura** | Redução de custo/latência com duas camadas de cache | -| ⚡**Solicitar Idempotência** | Janela de proteção duplicada | -| 🔒**Falsificação de impressão digital TLS** | Impressão digital TLS semelhante a navegador —**reduz a detecção de bots e a sinalização de contas** | -| 🔏**Correspondência de impressão digital CLI** | Corresponde às assinaturas de solicitação CLI nativas —**reduz o risco de banimento enquanto preserva o IP do proxy** | -| 🌐**Filtragem de IP** | Controle de lista de permissões/lista de bloqueio para implantações expostas | -| 📊**Limites de taxas editáveis** | Limites configuráveis ​​em nível global/de provedor com persistência | -| 📉**Degradação graciosa** | Fallbacks de capacidade multicamadas protegendo as principais operações de gateway | -| 📜**Trilha de auditoria de configuração** | Rastreamento de alterações baseado em diferenças, evitando desvios operacionais com reversões simples | -| ⏳**Sincronização de saúde do provedor** | Monitoramento proativo de expiração de token acionando alertas antes de falhas de autorização | -| 🚪**Desativar automaticamente contas banidas** | Disjuntor operacional vedando automaticamente contas de token permanentemente bloqueadas | -| 🔑**Gerenciamento de chaves de API + escopo** | Emissão/rotação segura de chaves e controles de modelo/provedor | -| 👁️**Revelação da chave de API com escopo**🆕 | Recuperação opcional de chaves de API via `ALLOW_API_KEY_REVEAL` | -| 🛡️**`/modelos` protegidos** | Autenticação opcional e ocultação de provedor para catálogo de modelos | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Recurso | O que faz | -| ----------------------------------------- | ---------------------------------------------------------------------------- | ---------------------------- | -| 📝**Solicitação + Registro de Proxy** | Solicitação/resposta completa e registro de proxy | -| 📉**Registros detalhados transmitidos**🆕 | Reconstrói fluxos de carga útil SSE de forma limpa na UI | -| 📋**Painel de registros unificado** | Visualizações de solicitação, proxy, auditoria e console em uma página | -| 🔍**Solicitar Telemetria** | Latência p50/p95/p99 e rastreamento de solicitação | -| 🏥**Painel de saúde** | Tempo de atividade, estados de disjuntores, bloqueios, estatísticas de cache | -| 💰**Acompanhamento de custos** | Controles de orçamento e visibilidade de preços por modelo | -| 📈**Visualizações analíticas** | Insights de uso de modelo/provedor e visualizações de tendências | -| 🧪**Estrutura de Avaliação** | Teste de Golden Set com estratégias de jogo configuráveis ​​ | -| 📡**Diagnóstico ao vivo**🆕 | Desvio de cache semântico para testes combinados ao vivo precisos | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Recurso | O que faz | -| ----------------------------------------- | ------------------------------------------------------------------------------------------ | --------------------- | -| 🌐**Implante em qualquer lugar** | Ambientes Localhost, VPS, Docker, Cloud | -| 🚇**Túnel Cloudflare**🆕 | Integração do Quick Tunnel com um clique no painel | -| 🔑**Filtragem de modelo de chave de API** | Resposta nativa /v1/models filtrada por meio de funções de contexto de portador atribuídas | -| ⚡**Ignorar Cache Inteligente** | Heurísticas TTL configuráveis ​​e controles de busca forçada | -| 🔄**Backup/Restauração** | Fluxos de exportação/importação e recuperação de desastres | -| 🧙**Assistente de integração** | Configuração guiada na primeira execução | -| 🔧**Painel de Ferramentas CLI** | Configuração com um clique para ferramentas de codificação populares | -| 🎮**Parque Modelo** | Teste qualquer provedor/modelo/endpoint no painel | -| 🔏**Alternar impressão digital CLI** | Correspondência de impressão digital por provedor em Configurações > Segurança | -| 🌐**i18n (30 idiomas)** | Painel completo + suporte a idiomas de documentos com cobertura RTL | -| 🧹**Limpar todos os modelos** | Limpeza da lista de modelos com um clique nos detalhes do fornecedor | -| 👁️**Controles da barra lateral**🆕 | Ocultar componentes e integrações em Configurações de aparência | -| 📋**Modelos de problemas** | Modelos padronizados do GitHub para bugs e recursos | -| 📂**Diretório de dados personalizado** | Substituição `DATA_DIR` para local de armazenamento | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1292,103 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -Quando a cota, a taxa ou a integridade falham, o OmniRoute passa automaticamente para o próximo candidato sem alternância manual.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A podem ser descobertos na interface do usuário e nos documentos (não ocultos) -- APIs de status de protocolo expõem dados operacionais em tempo real (`/api/mcp/*`, `/api/a2a/*`) -- Os painéis incluem ações para operações do dia 2 (alternâncias de combinação, reinicializações de disjuntores, cancelamento de tarefas)#### Translator + validation workflow +#### Protocol management that is visible and operable -A área do Tradutor inclui: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Playground**: solicita verificações de transformação -**Testador de bate-papo**: solicitação/resposta completa, ida e volta -**Banco de testes**: vários casos em uma execução -**Monitoramento ao vivo**: visualização do tráfego em tempo real +#### Translator + validation workflow -Além de validação de protocolo com clientes reais via `npm run test:protocols:e2e`. +The Translator area includes: -> 📖**[README do servidor MCP](open-sse/mcp-server/README.md)**— Referência de ferramentas, configurações de IDE e exemplos de clientes +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[README do servidor A2A](src/lib/a2a/README.md)**— Habilidades, métodos JSON-RPC, streaming e ciclo de vida de tarefas## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute inclui uma estrutura de avaliação integrada para testar a qualidade da resposta do LLM em relação a um conjunto dourado. Acesse-o em**Analytics → Evals**no painel.### Built-in Golden Set +## 🧪 Evaluations (Evals) -O "OmniRoute Golden Set" pré-carregado contém casos de teste para: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Saudações, matemática, geografia, geração de código -- Conformidade com o formato JSON, tradução, geração de descontos -- Recusa de segurança (conteúdo prejudicial), contagem, lógica booleana### Evaluation Strategies +### Built-in Golden Set -| Estratégia | Descrição | Exemplo | -| --------------- | --------------------------------------------------------------------------- | ----------------------------------- | --- | -| `exato` | A saída deve corresponder exatamente | `"4"` | -| `contém` | A saída deve conter substring (sem distinção entre maiúsculas e minúsculas) | `"Paris"` | -| `regex` | A saída deve corresponder ao padrão regex | `"1.*2.*3"` | -| `personalizado` | Função JS personalizada retorna verdadeiro/falso | `(saída) => saída.comprimento > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) - -🧩 Configuração do MCP (Protocolo de Contexto do Modelo) +
+🧩 MCP Setup (Model Context Protocol) -Inicie o transporte MCP no modo stdio:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Fluxo de validação recomendado: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Conecte seu cliente MCP por stdio. -2. Execute `omniroute_get_health`. -3. Execute `omniroute_list_combos`. -4. Abra `/dashboard/mcp` para confirmar pulsação, atividade e auditoria. - -APIs úteis para automação: +Useful APIs for automation: - `GET /api/mcp/status` - `GET /api/mcp/tools` -- `GET /api/mcp/auditoria` -- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit` +- `GET /api/mcp/audit/stats` - -🤝 Configuração A2A (Agent2Agent) + -Conheça o agente:```bash +
+🤝 A2A Setup (Agent2Agent) + +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Envie uma tarefa:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` - -Gerenciar ciclo de vida: +Manage lifecycle: - `GET /api/a2a/status` -- `GET /api/a2a/tarefas` +- `GET /api/a2a/tasks` - `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -IU operacional: +Operational UI: -- `/dashboard/a2a` para observabilidade de tarefa/estado/fluxo e ações de fumaça
+- `/dashboard/a2a` for task/state/stream observability and smoke actions - -🧪 Validação de protocolo ponta a ponta + -Valide ambos os protocolos com clientes reais:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Isso verifica: +This verifies: -- Conexão/lista/chamada do cliente MCP SDK -- Descoberta A2A/enviar/transmitir/obter/cancelar -- Verificação cruzada de dados em APIs de auditoria MCP e gerenciamento de tarefas A2A
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs - -💳 Provedores de assinatura### Claude Code (Pro/Max) + + +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1401,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Dica profissional:**Use o Opus para tarefas complexas e o Sonnet para velocidade. OmniRoute rastreia cota por modelo!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1415,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Cada conta do Codex agora possui opções de política em `Dashboard -> Providers`: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (ON/OFF): impõe a política de limite de janela de 5 horas. -- `Semanal` (LIGADO/DESLIGADO): impõe a política de limite de janela semanal. -- Comportamento do limite: quando uma janela habilitada atinge >=90% de uso, essa conta é ignorada. -- Comportamento de rotação: OmniRoute roteia automaticamente para a próxima conta Codex qualificada. -- Comportamento de redefinição: quando o tempo `resetAt` do provedor passa, a conta se torna elegível novamente automaticamente. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Cenários: +Scenarios: -- `5h ON` + `Weekly ON`: a conta é ignorada quando qualquer uma das janelas atinge o limite. -- `5h OFF` + `Weekly ON`: apenas o uso semanal pode bloquear a conta. -- `5h ON` + `Weekly OFF`: apenas o uso de 5 horas pode bloquear a conta. -- `resetAt` aprovado: a conta entra novamente na rotação automaticamente (sem reativação manual).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1440,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Melhor valor:**Grande nível gratuito! Use isso antes dos níveis pagos.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1455,71 +1662,91 @@ Models:
- -🔑 Provedores de chaves de API### NVIDIA NIM (FREE developer access — 70+ models) +
+🔑 API Key Providers -1. Cadastre-se: [build.nvidia.com](https://build.nvidia.com) -2. Obtenha uma chave de API gratuita (1.000 créditos de inferência incluídos) -3. Painel → Adicionar Provedor → NVIDIA NIM: - - Chave API: `nvapi-sua-chave` +### NVIDIA NIM (FREE developer access — 70+ models) -**Modelos:**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` e mais de 50 +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Dica profissional:**API compatível com OpenAI — funciona perfeitamente com a tradução de formato do OmniRoute!### DeepSeek +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -1. Cadastre-se: [platform.deepseek.com](https://platform.deepseek.com) -2. Obtenha a chave API -3. Painel → Adicionar provedor → DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -**Modelos:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +### DeepSeek -1. Cadastre-se: [console.groq.com](https://console.groq.com) -2. Obtenha a chave API (nível gratuito incluído) -3. Painel → Adicionar Provedor → Groq +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -**Modelos:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Dica profissional:**Inferência ultrarrápida — melhor para codificação em tempo real!### OpenRouter (100+ Models) +### Groq (Free Tier Available!) -1. Cadastre-se: [openrouter.ai](https://openrouter.ai) -2. Obtenha a chave API -3. Painel → Adicionar Provedor → OpenRouter +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -**Modelos:**acesse mais de 100 modelos de todos os principais fornecedores por meio de uma única chave de API. +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Comportamento do painel:**os modelos do OpenRouter são gerenciados a partir de**Modelos disponíveis**. Adição manual, importação e sincronização automática atualizam a mesma lista.
+**Pro Tip:** Ultra-fast inference — best for real-time coding! - -💰 Provedores baratos (backup)### GLM-4.7 (Daily reset, $0.6/1M) +### OpenRouter (100+ Models) -1. Inscreva-se: [Zhipu AI](https://open.bigmodel.cn/) -2. Obtenha a chave API do plano de codificação -3. Painel → Adicionar chave API: - - Provedor: `glm` - - Chave API: `sua-chave` +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -**Usar:**`glm/glm-4.7` +**Models:** Access 100+ models from all major providers through a single API key. -**Dica profissional:**O plano de codificação oferece cota 3× com custo de 1/7! Redefinir diariamente às 10h.### MiniMax M2.1 (5h reset, $0.20/1M) +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -1. Cadastre-se: [MiniMax](https://www.minimax.io/) -2. Obtenha a chave API -3. Painel → Adicionar chave API + -**Usar:**`minimax/MiniMax-M2.1` +
+💰 Cheap Providers (Backup) -**Dica profissional:**Opção mais barata para contexto longo (1 milhão de tokens)!### Kimi K2 ($9/month flat) +### GLM-4.7 (Daily reset, $0.6/1M) -1. Inscreva-se: [Moonshot AI](https://platform.moonshot.ai/) -2. Obtenha a chave API -3. Painel → Adicionar chave API +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**Usar:**`kimi/kimi-latest` +**Use:** `glm/glm-4.7` -**Dica profissional:**$9 fixos/mês para 10 milhões de tokens = $0,90/custo efetivo de 1 milhão!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. - -🆓 Provedores GRATUITOS (backup de emergência)### Qoder (5 FREE models via OAuth) +### MiniMax M2.1 (5h reset, $0.20/1M) + +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `minimax/MiniMax-M2.1` + +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1560,8 +1787,10 @@ Models:
- -🎨 Criar Combos### Example 1: Maximize Subscription → Cheap Backup +
+🎨 Create Combos + +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1589,8 +1818,10 @@ Cost: $0 forever!
- -🔧 Integração CLI### Cursor IDE +
+🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1601,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Use a página**Ferramentas CLI**no painel para configuração com um clique ou edite `~/.claude/settings.json` manualmente.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1612,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Opção 1 — Painel (recomendado):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Opção 2 — Manual:**Edite `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1629,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Observação:**OpenClaw só funciona com OmniRoute local. Use `127.0.0.1` em vez de `localhost` para evitar problemas de resolução IPv6.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1643,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Etapa 1:**Adicione OmniRoute como um provedor personalizado:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Etapa 2:**Crie/edite `opencode.json` na raiz do seu projeto:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1669,117 +1909,130 @@ opencode } } } -```` +``` -**Etapa 3:**Selecione o modelo no OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Dica:**Adicione qualquer modelo disponível no endpoint `/v1/models` do OmniRoute à seção `models`. Use o formato `provider/model-id` do painel do OmniRoute.
+ --- ## Resolução de Problemas - -Clique para expandir o guia de solução de problemas +
+Click to expand troubleshooting guide -**"O modelo de linguagem não forneceu mensagens"** +**"Language model did not provide messages"** -- Cota do provedor esgotada → Verifique o rastreador de cota do painel -- Solução: use o combo substituto ou mude para um nível mais barato +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**Limitação de taxa** +**Rate limiting** -- Cota de assinatura esgotada → Fallback para GLM/MiniMax -- Adicionar combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**O token OAuth expirou** +**OAuth token expired** -- Atualizado automaticamente pelo OmniRoute -- Se os problemas persistirem: Painel → Provedor → Reconectar +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Custos elevados** +**High costs** -- Verifique as estatísticas de uso em Painel → Custos -- Mude o modelo primário para GLM/MiniMax -- Use o nível gratuito (Gemini CLI, Qoder) para tarefas não críticas +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**As portas do painel/API estão erradas** +**Dashboard/API ports are wrong** -- `PORT` é a porta base canônica (e porta API por padrão) -- `API_PORT` substitui apenas o ouvinte de API compatível com OpenAI -- `DASHBOARD_PORT` substitui apenas o ouvinte dashboard/Next.js -- Defina `NEXT_PUBLIC_BASE_URL` como seu painel/URL público (para retornos de chamada OAuth) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Erros de sincronização na nuvem** +**Cloud sync errors** -- Verifique se `BASE_URL` aponta para sua instância em execução -- Verifique os pontos `CLOUD_URL` para o endpoint de nuvem esperado -- Mantenha os valores `NEXT_PUBLIC_*` alinhados com os valores do lado do servidor +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Primeiro login não funciona** +**First login not working** -- Verifique `INITIAL_PASSWORD` em `.env` -- Se não definida, a senha substituta é `123456` +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Sem registros de solicitação** +**No request logs** -- Os artefatos de solicitação são gravados em `DATA_DIR/call_logs/` como um arquivo JSON por solicitação -- Habilite a captura de pipeline em Painel → Logs → Solicitar logs se precisar de cargas úteis detalhadas por estágio -- Defina `APP_LOG_TO_FILE=true` se você também deseja logs do console do aplicativo em `logs/application/app.log` -- Ajuste `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` e `CALL_LOG_MAX_ENTRIES` conforme necessário +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**O teste de conexão mostra "Inválido" para provedores compatíveis com OpenAI** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Muitos provedores não expõem um endpoint `/models` -- OmniRoute v1.0.6+ inclui validação de fallback por meio de conclusões de chat -- Certifique-se de que o URL base inclua o sufixo `/v1`### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ Importante para usuários executando OmniRoute em um VPS, Docker ou qualquer servidor remoto**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -Os provedores**Antigravity**e**Gemini CLI**usam o**Google OAuth 2.0**. O Google exige que `redirect_uri` no fluxo OAuth corresponda exatamente a um dos URIs pré-registrados no Google Cloud Console do aplicativo. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -As credenciais OAuth incluídas no OmniRoute são registradas**apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (por exemplo, `https://omniroute.myserver.com`), o Google rejeita a autenticação com:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Você precisa criar um**ID do cliente OAuth 2.0**no Console do Google Cloud com o URI do seu servidor.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Abra o Console do Google Cloud** +#### Step-by-step -Acesse: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Crie um novo ID de cliente OAuth 2.0** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Clique em**"+ Criar credenciais"**→**"ID do cliente OAuth"** -- Tipo de aplicativo:**"Aplicativo Web"** -- Nome: o que você quiser (por exemplo, `OmniRoute Remote`) +**2. Create a new OAuth 2.0 Client ID** -**3. Adicionar URIs de redirecionamento autorizados** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -No campo**"URIs de redirecionamento autorizados"**, adicione:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Substitua `your-server.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, por exemplo, `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. Salve e copie as credenciais** +After creating, Google will show the **Client ID** and **Client Secret**. -Após a criação, o Google mostrará o**ID do cliente**e o**Segredo do cliente**. +**5. Set environment variables** -**5. Definir variáveis de ambiente** +In your `.env` (or Docker environment variables): -No seu `.env` (ou variáveis de ambiente do Docker):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1788,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Reinicie o OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Tente conectar novamente** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Painel → Provedores → Antigravidade (ou Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -O Google agora redirecionará corretamente para `https://your-server.com/callback`.--- +--- #### Temporary workaround (without custom credentials) -Se não quiser configurar suas próprias credenciais agora, você ainda pode usar o**fluxo manual de URL**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute abre o URL de autorização do Google -2. Após autorização, o Google tenta redirecionar para `localhost` (que falha no servidor remoto) -3.**Copie o URL completo**da barra de endereço do seu navegador (mesmo que a página não carregue) -4. Cole esse URL no campo mostrado no modal de conexão OmniRoute -5. Clique em**"Conectar"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Isso funciona porque o código de autorização no URL é válido independentemente de a página de redirecionamento ter sido carregada.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. - -🇧🇷 Versão em Português#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Os provedores**Antigravity**e**Gemini CLI**usam**Google OAuth 2.0**para autenticação. O Google exige que um `redirect_uri` usado no fluxo OAuth seja**exatamente**uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. +
+🇧🇷 Versão em Português -As credenciais OAuth incorporadas no OmniRoute estão cadastradas**apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Você precisa criar um**OAuth 2.0 Client ID**no Google Cloud Console com o URI do seu servidor.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Acesse o Console do Google Cloud** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -**2. Crie um novo ID de cliente OAuth 2.0** +**2. Crie um novo OAuth 2.0 Client ID** -- Clique em**"+ Criar credenciais"**→**"ID do cliente OAuth"** -- Tipo de aplicativo:**"Aplicativo Web"** +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** - Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Adicionar como URIs de redirecionamento autorizados** +**3. Adicione as Authorized Redirect URIs** -No campo**"URIs de redirecionamento autorizados"**, adicionado:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback - -```` +``` > Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). **4. Salve e copie as credenciais** -Após criar, o Google mostrará o**Client ID**e o**Client Secret**. +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -**5. Configurar como variáveis de ambiente** +**5. Configure as variáveis de ambiente** -No seu `.env` (ou nas variáveis de ambiente do Docker):```bash +No seu `.env` (ou nas variáveis de ambiente do Docker): + +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1867,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reinicie o OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute - -```` +``` **7. Tente conectar novamente** -Painel → Provedores → Antigravidade (ou Gemini CLI) → OAuth +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará.--- +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. + +--- #### Workaround temporário (sem configurar credenciais próprias) -Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo**manual de URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. O OmniRoute abrirá uma URL de autorização do Google +1. O OmniRoute abrirá a URL de autorização do Google 2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) -3.**Copie a URL completa**da barra de endereço do seu navegador (mesmo que a página não carregue) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) 4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute -5. Clique em**"Conectar"** +5. Clique em **"Connect"** -> Esta solução alternativa funciona porque o código de autorização na URL é válido, independentemente do redirecionamento ter sido carregado ou não.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1905,64 +2171,73 @@ Se não quiser criar credenciais próprias agora, ainda é possível usar o flux ## 🛠️ Tech Stack - -Clique para expandir os detalhes da pilha de tecnologia +
+Click to expand tech stack details --**Tempo de execução**: Node.js 18–22 LTS (⚠️ Node.js 24+**não é compatível**— binários nativos `better-sqlite3` são incompatíveis) --**Idioma**: TypeScript 5.9 —**100% TypeScript**em `src/` e `open-sse/` (zero `any` nos módulos principais desde a v2.0) --**Estrutura**: Next.js 16 + React 19 + Tailwind CSS 4 --**Banco de dados**: LowDB (JSON) + SQLite (estado do domínio + logs de proxy + auditoria MCP + decisões de roteamento) --**Esquemas**: Zod (validação de E/S da ferramenta MCP, contratos de API) --**Protocolos**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Streaming**: eventos enviados pelo servidor (SSE) --**Auth**: OAuth 2.0 (PKCE) + JWT + Chaves de API + Autorização com escopo MCP --**Testes**: executor de testes Node.js + Vitest (mais de 900 testes incluindo unidade, integração, E2E) --**CI/CD**: GitHub Actions (publicação automática de npm + Docker Hub no lançamento) --**Site**: [omniroute.online](https://omniroute.online) --**Pacote**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Resiliência**: Disjuntor, espera exponencial, rebanho anti-trovão, falsificação de TLS, autocura de combinação automática
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Documentação -| Documento | Descrição | -| -------------------------------------------------------- | --------------------------------------------------- | -| [Guia do usuário](docs/USER_GUIDE.md) | Provedores, combos, integração CLI, implantação | -| [Referência de API](docs/API_REFERENCE.md) | Todos os endpoints com exemplos | -| [Servidor MCP](open-sse/mcp-server/README.md) | 16 ferramentas MCP, configurações IDE, clientes Python/TS/Go | -| [Servidor A2A](src/lib/a2a/README.md) | Protocolo JSON-RPC 2.0, habilidades, streaming, gerenciamento de tarefas | -| [Mecanismo de combinação automática](docs/auto-combo.md) | Pontuação de 6 fatores, pacotes de modos, autocura | -| [Solução de problemas](docs/TROUBLESHOOTING.md) | Problemas e soluções comuns | -| [Arquitetura](docs/ARCHITECTURE.md) | Arquitetura do sistema e componentes internos | -| [Contribuindo](CONTRIBUTING.md) | Configuração e diretrizes de desenvolvimento | -| [Especificações OpenAPI](docs/openapi.yaml) | Especificação OpenAPI 3.0 | -| [Política de Segurança](SECURITY.md) | Relatórios de vulnerabilidades e práticas de segurança | -| [Implantação de VM](docs/VM_DEPLOYMENT_GUIDE.md) | Guia completo: configuração de VM + nginx + Cloudflare | -| [Galeria de recursos](docs/FEATURES.md) | Tour visual do painel com capturas de tela | -| [Lista de verificação de lançamento](docs/RELEASE_CHECKLIST.md) | Etapas de validação de pré-lançamento |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute tem**210+ recursos planejados**em diversas fases de desenvolvimento. Aqui estão as principais áreas: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Categoria | Recursos planejados | Destaques | +| Category | Planned Features | Highlights | | ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | -| 🧠**Roteamento e Inteligência**| 25+ | Roteamento de menor latência, roteamento baseado em tags, simulação de cota, seleção de conta P2C | -| 🔒**Segurança e Conformidade**| 20+ | Proteção SSRF, camuflagem de credenciais, limite de taxa por endpoint, escopo de chave de gerenciamento | -| 📊**Observabilidade**| 15+ | Integração OpenTelemetry, monitoramento de cotas em tempo real, rastreamento de custos por modelo | -| 🔄**Integrações com Provedores**| 20+ | Registro de modelo dinâmico, resfriamento de provedor, Codex multicontas, análise de cotas do Copilot | -| ⚡**Desempenho**| 15+ | Camada de cache dupla, cache de prompt, cache de resposta, manutenção de atividade de streaming, API em lote | -| 🌐**Ecossistema**| 10+ | API WebSocket, configuração hot-reload, armazenamento de configuração distribuído, modo comercial |### 🔜 Coming Soon +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**Integração OpenCode**— Suporte de provedor nativo para o IDE de codificação OpenCode AI -- 🔗**Integração TRAE**— Suporte total para a estrutura de desenvolvimento TRAE AI -- 📦**API Batch**— Processamento assíncrono em lote para solicitações em massa -- 🎯**Roteamento baseado em tags**— Roteie solicitações com base em tags personalizadas e metadados -- 💰**Estratégia de custo mais baixo**— Selecione automaticamente o provedor mais barato disponível +### 🔜 Coming Soon -> 📝 Especificações completas de recursos disponíveis em [`docs/new-features/`](docs/new-features/) (217 especificações detalhadas)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1970,18 +2245,20 @@ OmniRoute tem**210+ recursos planejados**em diversas fases de desenvolvimento. A ### How to Contribute -1. Bifurque o repositório -2. Crie seu branch de recursos (`git checkout -b feature/amazing-feature`) -3. Confirme suas alterações (`git commit -m 'Adicionar recurso incrível'`) -4. Envie para o branch (`git push origin feature/amazing-feature`) -5. Abra uma solicitação pull +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Consulte [CONTRIBUTING.md](CONTRIBUTING.md) para obter diretrizes detalhadas.### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1993,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Agradecimentos especiais a**[9router](https://github.com/decolua/9router)**de**[decolua](https://github.com/decolua)**— o projeto original que inspirou este fork. OmniRoute se baseia nessa base incrível com recursos adicionais, APIs multimodais e uma reescrita completa do TypeScript. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Agradecimentos especiais a**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— a implementação Go original que inspirou esta versão JavaScript.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Licença -Licença MIT - consulte [LICENSE](LICENSE) para obter detalhes.--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/pt/docs/ARCHITECTURE.md b/docs/i18n/pt/docs/ARCHITECTURE.md index e1daecdfbd..dac4ff9094 100644 --- a/docs/i18n/pt/docs/ARCHITECTURE.md +++ b/docs/i18n/pt/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Última atualização: 28/03/2026_## Executive Summary -OmniRoute é um gateway de roteamento de IA local e painel construído em Next.js. -Ele fornece um único endpoint compatível com OpenAI (`/v1/*`) e roteia o tráfego entre vários provedores upstream com tradução, fallback, atualização de token e rastreamento de uso. -Capacidades principais: +_Last updated: 2026-03-28_ -- Superfície API compatível com OpenAI para CLI/ferramentas (28 provedores) -- Tradução de solicitação/resposta em formatos de provedores -- Fallback de combinação de modelos (sequência de vários modelos) -- Fallback em nível de conta (várias contas por provedor) -- Gerenciamento de conexão de provedor de chave OAuth + API -- Geração de incorporação via `/v1/embeddings` (6 provedores, 9 modelos) -- Geração de imagens via `/v1/images/Generations` (4 provedores, 9 modelos) -- Pense na análise de tags (`...`) para modelos de raciocínio -- Sanitização de resposta para compatibilidade estrita com OpenAI SDK -- Normalização de funções (desenvolvedor→sistema, sistema→usuário) para compatibilidade entre provedores -- Conversão de saída estruturada (json_schema → Gemini responseSchema) -- Persistência local para provedores, chaves, aliases, combos, configurações, preços -- Acompanhamento de uso/custo e registro de solicitações -- Sincronização em nuvem opcional para sincronização de vários dispositivos/estado -- Lista de permissões/lista de bloqueio de IP para controle de acesso à API -- Pensando na gestão orçamentária (passthrough/auto/custom/adaptive) -- Injeção imediata do sistema global -- Rastreamento de sessão e impressão digital -- Limitação de taxa aprimorada por conta com perfis específicos do provedor -- Padrão de disjuntor para resiliência do provedor -- Proteção de rebanho anti-trovão com bloqueio mutex -- Cache de desduplicação de solicitação baseada em assinatura -- Camada de domínio: disponibilidade do modelo, regras de custo, política de fallback, política de bloqueio -- Persistência de estado de domínio (cache write-through SQLite para fallbacks, orçamentos, bloqueios, disjuntores) -- Mecanismo de política para avaliação centralizada de solicitações (bloqueio → orçamento → fallback) -- Solicitar telemetria com agregação de latência p50/p95/p99 -- ID de correlação (X-Request-Id) para rastreamento ponta a ponta -- Registro de auditoria de conformidade com cancelamento por chave de API -- Estrutura de avaliação para garantia de qualidade LLM -- Painel de UI de resiliência com status do disjuntor em tempo real -- Provedores OAuth modulares (12 módulos individuais em `src/lib/oauth/providers/`) +## Executive Summary -Modelo de tempo de execução primário: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- As rotas do aplicativo Next.js em `src/app/api/*` implementam APIs de painel e APIs de compatibilidade -- Um núcleo SSE/roteamento compartilhado em `src/sse/*` + `open-sse/*` lida com execução, tradução, streaming, fallback e uso do provedor## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Tempo de execução do gateway local -- APIs de gerenciamento de painel -- Autenticação do provedor e atualização de token -- Solicitar tradução e streaming SSE -- Estado local + persistência de uso -- Orquestração opcional de sincronização em nuvem### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Implementação de serviço em nuvem por trás de `NEXT_PUBLIC_CLOUD_URL` -- Plano de controle/SLA do provedor fora do processo local -- Os próprios binários CLI externos (Claude CLI, Codex CLI, etc.)## Dashboard Surface (Current) +### Out of Scope -Páginas principais em `src/app/(dashboard)/dashboard/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — início rápido + visão geral do provedor -- `/dashboard/endpoint` — proxy de endpoint + MCP + A2A + guias de endpoint de API -- `/dashboard/providers` — conexões e credenciais do provedor -- `/dashboard/combos` — estratégias de combinação, modelos, regras de roteamento de modelo -- `/dashboard/costs` — agregação de custos e visibilidade de preços -- `/dashboard/analytics` — análises e avaliações de uso -- `/dashboard/limits` — controles de cota/taxa -- `/dashboard/cli-tools` — integração CLI, detecção de tempo de execução, geração de configuração -- `/dashboard/agents` — agentes ACP detectados + registro de agente personalizado -- `/dashboard/media` — playground de imagem/vídeo/música -- `/dashboard/search-tools` — teste e histórico do provedor de pesquisa -- `/dashboard/health` — tempo de atividade, disjuntores, limites de taxa -- `/dashboard/logs` — solicitação/proxy/auditoria/logs do console -- `/dashboard/settings` — guias de configurações do sistema (geral, roteamento, padrões de combinação, etc.) -- `/dashboard/api-manager` — Ciclo de vida da chave de API e permissões de modelo## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Diretórios principais: +Main directories: -- `src/app/api/v1/*` e `src/app/api/v1beta/*` para APIs de compatibilidade -- `src/app/api/*` para APIs de gerenciamento/configuração -- Próximas reescritas em `next.config.mjs` mapeiam `/v1/*` para `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Rotas de compatibilidade importantes: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` — inclui modelos personalizados com `custom: true` -- `src/app/api/v1/embeddings/route.ts` — geração de incorporação (6 provedores) -- `src/app/api/v1/images/generations/route.ts` — geração de imagens (4+ provedores incluindo Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — bate-papo dedicado por provedor -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — embeddings dedicados por provedor -- `src/app/api/v1/providers/[provider]/images/Generations/route.ts` — imagens dedicadas por provedor +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` -- `src/app/api/v1beta/models/[...caminho]/route.ts` +- `src/app/api/v1beta/models/[...path]/route.ts` -Domínios de gerenciamento: +Management domains: -- Autenticação/configurações: `src/app/api/auth/*`, `src/app/api/settings/*` -- Provedores/conexões: `src/app/api/providers*` -- Nós do provedor: `src/app/api/provider-nodes*` -- Modelos personalizados: `src/app/api/provider-models` (GET/POST/DELETE) -- Catálogo de modelos: `src/app/api/models/route.ts` (GET) -- Configuração de proxy: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- Chaves/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- Uso: `src/app/api/usage/*` -- Sincronização/nuvem: `src/app/api/sync/*`, `src/app/api/cloud/*` -- Auxiliares de ferramentas CLI: `src/app/api/cli-tools/*` -- Filtro IP: `src/app/api/settings/ip-filter` (GET/PUT) -- Pensando no orçamento: `src/app/api/settings/thinking-budget` (GET/PUT) -- Prompt do sistema: `src/app/api/settings/system-prompt` (GET/PUT) -- Sessões: `src/app/api/sessions` (GET) -- Limites de taxa: `src/app/api/rate-limits` (GET) -- Resiliência: `src/app/api/resilience` (GET/PATCH) — perfis de provedor, disjuntor, estado limite de taxa -- Redefinição de resiliência: `src/app/api/resilience/reset` (POST) — redefinir disjuntores + resfriamento -- Estatísticas de cache: `src/app/api/cache/stats` (GET/DELETE) -- Disponibilidade do modelo: `src/app/api/models/availability` (GET/POST) -- Telemetria: `src/app/api/telemetry/summary` (GET) -- Orçamento: `src/app/api/usage/budget` (GET/POST) -- Cadeias de fallback: `src/app/api/fallback/chains` (GET/POST/DELETE) -- Auditoria de conformidade: `src/app/api/compliance/audit-log` (GET) -- Avaliações: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- Políticas: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -Principais módulos de fluxo: +## 2) SSE + Translation Core -- Entrada: `src/sse/handlers/chat.ts` -- Orquestração central: `open-sse/handlers/chatCore.ts` -- Adaptadores de execução do provedor: `open-sse/executors/*` -- Detecção de formato/configuração do provedor: `open-sse/services/provider.ts` -- Análise/resolução de modelo: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Lógica de fallback de conta: `open-sse/services/accountFallback.ts` -- Registro de tradução: `open-sse/translator/index.ts` -- Transformações de fluxo: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- Extração/normalização de uso: `open-sse/utils/usageTracking.ts` -- Pense no analisador de tags: `open-sse/utils/thinkTagParser.ts` -- Manipulador de incorporação: `open-sse/handlers/embeddings.ts` -- Registro do provedor de incorporação: `open-sse/config/embeddingRegistry.ts` -- Manipulador de geração de imagem: `open-sse/handlers/imageGeneration.ts` -- Registro do provedor de imagens: `open-sse/config/imageRegistry.ts` -- Sanitização de resposta: `open-sse/handlers/responseSanitizer.ts` -- Normalização de funções: `open-sse/services/roleNormalizer.ts` +Main flow modules: -Serviços (lógica de negócios): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- Seleção/pontuação de conta: `open-sse/services/accountSelector.ts` -- Gerenciamento do ciclo de vida do contexto: `open-sse/services/contextManager.ts` -- Aplicação do filtro IP: `open-sse/services/ipFilter.ts` -- Rastreamento de sessão: `open-sse/services/sessionManager.ts` -- Solicitar desduplicação: `open-sse/services/signatureCache.ts` -- Injeção de prompt do sistema: `open-sse/services/systemPrompt.ts` -- Pensando na gestão orçamentária: `open-sse/services/thinkingBudget.ts` -- Roteamento de modelo curinga: `open-sse/services/wildcardRouter.ts` -- Gerenciamento de limite de taxa: `open-sse/services/rateLimitManager.ts` -- Disjuntor: `open-sse/services/circuitBreaker.ts` +Services (business logic): -Módulos da camada de domínio: +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- Disponibilidade do modelo: `src/lib/domain/modelAvailability.ts` -- Regras/orçamentos de custo: `src/lib/domain/costRules.ts` -- Política de fallback: `src/lib/domain/fallbackPolicy.ts` -- Resolvedor combinado: `src/lib/domain/comboResolver.ts` -- Política de bloqueio: `src/lib/domain/lockoutPolicy.ts` -- Mecanismo de política: `src/domain/policyEngine.ts` — bloqueio centralizado → orçamento → avaliação de fallback -- Catálogo de códigos de erro: `src/lib/domain/errorCodes.ts` -- ID da solicitação: `src/lib/domain/requestId.ts` -- Tempo limite de busca: `src/lib/domain/fetchTimeout.ts` -- Solicitar telemetria: `src/lib/domain/requestTelemetry.ts` -- Conformidade/auditoria: `src/lib/domain/compliance/index.ts` -- Corredor de avaliação: `src/lib/domain/evalRunner.ts` -- Persistência de estado de domínio: `src/lib/db/domainState.ts` — SQLite CRUD para cadeias de fallback, orçamentos, histórico de custos, estado de bloqueio, disjuntores +Domain layer modules: -Módulos do provedor OAuth (12 arquivos individuais em `src/lib/oauth/providers/`): +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -- Índice de registro: `src/lib/oauth/providers/index.ts` -- Provedores individuais: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` -- Thin wrapper: `src/lib/oauth/providers.ts` — reexportações de módulos individuais## 3) Persistence Layer +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -Banco de dados de estado primário (SQLite): +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -- Infra principal: `src/lib/db/core.ts` (better-sqlite3, migrações, WAL) -- Fachada de reexportação: `src/lib/localDb.ts` (camada de compatibilidade fina para chamadores) -- arquivo: `${DATA_DIR}/storage.sqlite` (ou `$XDG_CONFIG_HOME/omniroute/storage.sqlite` quando definido, senão `~/.omniroute/storage.sqlite`) -- entidades (tabelas + namespaces KV): ProviderConnections, ProvideNodes, modelAliases, combos, apiKeys, configurações, preços,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +## 3) Persistence Layer -Persistência de uso: +Primary state DB (SQLite): -- fachada: `src/lib/usageDb.ts` (módulos decompostos em `src/lib/usage/*`) -- Tabelas SQLite em `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` -- artefatos de arquivo opcionais permanecem para compatibilidade/depuração (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- arquivos JSON legados são migrados para SQLite por migrações de inicialização, quando presentes +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -Banco de dados de estado de domínio (SQLite): +Usage persistence: -- `src/lib/db/domainState.ts` — operações CRUD para estado de domínio -- Tabelas (criadas em `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- Padrão de cache write-through: os mapas na memória são autoritativos em tempo de execução; as mutações são escritas de forma síncrona no SQLite; o estado é restaurado do banco de dados na inicialização a frio## 4) Auth + Security Surfaces +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- Autenticação de cookie do painel: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- Geração/verificação de chave de API: `src/shared/utils/apiKey.ts` -- Os segredos do provedor persistiram nas entradas `providerConnections` -- Suporte a proxy de saída via `open-sse/utils/proxyFetch.ts` (env vars) e `open-sse/utils/networkProxy.ts` (configurável por provedor ou global)## 5) Cloud Sync +Domain State DB (SQLite): -- Inicialização do agendador: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Tarefa periódica: `src/shared/services/cloudSyncScheduler.ts` -- Tarefa periódica: `src/shared/services/modelSyncScheduler.ts` -- Rota de controle: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -As decisões de fallback são conduzidas por `open-sse/services/accountFallback.ts` usando códigos de status e heurísticas de mensagens de erro. O roteamento combinado adiciona uma proteção extra: 400s no escopo do provedor, como falhas de bloqueio de conteúdo upstream e de validação de função, são tratadas como falhas locais do modelo para que destinos combinados posteriores ainda possam ser executados.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -A atualização durante o tráfego ao vivo é executada dentro de `open-sse/handlers/chatCore.ts` por meio do executor `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -A sincronização periódica é acionada por `CloudSyncScheduler` quando a nuvem está habilitada.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -Arquivos de armazenamento físico: +Physical storage files: -- banco de dados de tempo de execução primário: `${DATA_DIR}/storage.sqlite` -- solicitar linhas de log: `${DATA_DIR}/log.txt` (artefato de compatibilidade/depuração) -- arquivos de carga útil de chamada estruturada: `${DATA_DIR}/call_logs/` -- sessões opcionais de depuração de tradução/solicitação: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: APIs de compatibilidade -- `src/app/api/v1/providers/[provider]/*`: rotas dedicadas por provedor (chat, embeddings, imagens) -- `src/app/api/providers*`: provedor CRUD, validação, teste -- `src/app/api/provider-nodes*`: gerenciamento de nó personalizado compatível -- `src/app/api/provider-models`: gerenciamento de modelo personalizado (CRUD) -- `src/app/api/models/route.ts`: API de catálogo de modelos (aliases + modelos personalizados) -- `src/app/api/oauth/*`: fluxos OAuth/código do dispositivo -- `src/app/api/keys*`: ciclo de vida da chave de API local -- `src/app/api/models/alias`: gerenciamento de alias -- `src/app/api/combos*`: gerenciamento de combo substituto -- `src/app/api/pricing`: substituições de preços para cálculo de custos -- `src/app/api/settings/proxy`: configuração de proxy (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: teste de conectividade de proxy de saída (POST) -- `src/app/api/usage/*`: APIs de uso e logs -- `src/app/api/sync/*` + `src/app/api/cloud/*`: sincronização na nuvem e ajudantes voltados para a nuvem -- `src/app/api/cli-tools/*`: gravadores/verificadores de configuração CLI locais -- `src/app/api/settings/ip-filter`: lista de permissões/lista de bloqueios de IP (GET/PUT) -- `src/app/api/settings/thinking-budget`: configuração do orçamento do token de pensamento (GET/PUT) -- `src/app/api/settings/system-prompt`: prompt global do sistema (GET/PUT) -- `src/app/api/sessions`: listagem de sessões ativas (GET) -- `src/app/api/rate-limits`: status do limite de taxa por conta (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: análise de solicitação, manipulação de combo, loop de seleção de conta -- `open-sse/handlers/chatCore.ts`: tradução, envio do executor, manipulação de novas tentativas/atualizações, configuração de stream -- `open-sse/executors/*`: rede específica do provedor e comportamento do formato### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: registro e orquestração do tradutor -- Solicitar tradutores: `open-sse/translator/request/*` -- Tradutores de resposta: `open-sse/translator/response/*` -- Constantes de formato: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: configuração/estado persistente e persistência de domínio no SQLite -- `src/lib/localDb.ts`: reexportação de compatibilidade para módulos de banco de dados -- `src/lib/usageDb.ts`: histórico de uso/fachada de logs de chamadas em cima de tabelas SQLite## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Cada provedor tem um executor especializado que estende `BaseExecutor` (em `open-sse/executors/base.ts`), que fornece construção de URL, construção de cabeçalho, nova tentativa com espera exponencial, ganchos de atualização de credenciais e o método de orquestração `execute()`. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Executor | Fornecedor(es) | Tratamento Especial | -| --------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------- | -| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Juntos, Fireworks, Cerebras, Cohere, NVIDIA | Configuração dinâmica de URL/cabeçalho por provedor | -| `AntigravityExecutor` | Antigravidade do Google | IDs de projeto/sessão personalizados, análise repetida após | -| `CodexExecutor` | Códice OpenAI | Injeta instruções do sistema, força esforço de raciocínio | -| `CursorExecutor` | Cursor IDE | Protocolo ConnectRPC, codificação Protobuf, assinatura de solicitação via checksum | -| `GithubExecutor` | Copiloto GitHub | Atualização de token do copiloto, cabeçalhos que imitam VSCode | -| `KiroExecutor` | AWS CodeWhisperer/Kiro | Formato binário AWS EventStream → conversão SSE | -| `GeminiCLIExecutor` | Gêmeos CLI | Ciclo de atualização do token OAuth do Google | +### Persistence -Todos os outros provedores (incluindo nós compatíveis personalizados) usam o `DefaultExecutor`.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Provedor | Formato | Autenticação | Transmitir | Não-transmissão | Atualização de token | API de uso | -| ------------------------ | ---------------- | --------------------------------- | ---------------- | --------------- | -------------------- | ------------------------ | ------------------------------ | -| Cláudio | Cláudio | Chave API/OAuth | ✅ | ✅ | ✅ | ⚠️ Somente administrador | -| Gêmeos | gêmeos | Chave API/OAuth | ✅ | ✅ | ✅ | ⚠️Console em nuvem | -| Gêmeos CLI | gêmeo-cli | OAuth | ✅ | ✅ | ✅ | ⚠️Console em nuvem | -| Antigravidade | antigravidade | OAuth | ✅ | ✅ | ✅ | ✅ API de cota completa | -| OpenAI | abrirai | Chave API | ✅ | ✅ | ❌ | ❌ | -| Códice | respostas openai | OAuth | ✅ forçado | ❌ | ✅ | ✅ Limites de taxas | -| Copiloto GitHub | abrirai | OAuth + token de copiloto | ✅ | ✅ | ✅ | ✅ Instantâneos de cota | -| Cursor | cursor | Soma de verificação personalizada | ✅ | ✅ | ❌ | ❌ | -| Kiro | Kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Limites de uso | -| Qwen | abrirai | OAuth | ✅ | ✅ | ✅ | ⚠️ Por solicitação | -| Qoder | abrirai | OAuth (Básico) | ✅ | ✅ | ✅ | ⚠️ Por solicitação | -| OpenRouter | abrirai | Chave API | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | Cláudio | Chave API | ✅ | ✅ | ❌ | ❌ | -| DeepSeek | abrirai | Chave API | ✅ | ✅ | ❌ | ❌ | -| Groq | abrirai | Chave API | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | abrirai | Chave API | ✅ | ✅ | ❌ | ❌ | -| Mistral | abrirai | Chave API | ✅ | ✅ | ❌ | ❌ | -| Perplexidade | abrirai | Chave API | ✅ | ✅ | ❌ | ❌ | -| Juntos IA | abrirai | Chave API | ✅ | ✅ | ❌ | ❌ | -| IA de fogos de artifício | abrirai | Chave API | ✅ | ✅ | ❌ | ❌ | -| Cérebros | abrirai | Chave API | ✅ | ✅ | ❌ | ❌ | -| Coerente | abrirai | Chave API | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | abrirai | Chave API | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Os formatos de origem detectados incluem: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. + +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | + +All other providers (including custom compatible nodes) use the `DefaultExecutor`. + +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: - `openai` -- `openai-respostas` -- `cláudio` -- `gêmeos` +- `openai-responses` +- `claude` +- `gemini` -Os formatos de destino incluem: +Target formats include: -- Bate-papo/respostas OpenAI - -Cláudio -- Envelope Gemini/Gemini-CLI/Antigravidade - -Kiro +- OpenAI chat/Responses +- Claude +- Gemini/Gemini-CLI/Antigravity envelope +- Kiro - Cursor -As traduções usam**OpenAI como formato de hub**— todas as conversões passam pelo OpenAI como intermediário:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -As traduções são selecionadas dinamicamente com base no formato da carga útil de origem e no formato de destino do provedor. +Additional processing layers in the translation pipeline: -Camadas de processamento adicionais no pipeline de tradução: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Sanitização de respostas**— Remove campos não padrão de respostas no formato OpenAI (streaming e não streaming) para garantir conformidade estrita com o SDK --**Normalização de funções**— Converte `desenvolvedor` → `sistema` para alvos não-OpenAI; mescla `sistema` → `usuário` para modelos que rejeitam a função do sistema (GLM, ERNIE) --**Extração de tag Think**— Analisa blocos `...` do conteúdo no campo `reasoning_content` --**Saída estruturada**— Converte `response_format.json_schema` do OpenAI em `responseMimeType` + `responseSchema` do Gemini## Supported API Endpoints +## Supported API Endpoints -| Ponto final | Formato | Manipulador | -| -------------------------------------------------- | ------------------ | ---------------------------------------------------------------------------------- | -| `POST /v1/chat/completions` | Bate-papo OpenAI | `src/sse/handlers/chat.ts` | -| `POST /v1/mensagens` | Mensagens de Cláudio | Mesmo manipulador (detectado automaticamente) | -| `POST /v1/respostas` | Respostas OpenAI | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/embeddings` | Incorporações OpenAI | `open-sse/handlers/embeddings.ts` | -| `GET /v1/embeddings` | Listagem de modelos | Rota API | -| `POST /v1/imagens/gerações` | Imagens OpenAI | `open-sse/handlers/imageGeneration.ts` | -| `GET /v1/imagens/gerações` | Listagem de modelos | Rota API | -| `POST /v1/provedores/{provedor}/chat/completions` | Bate-papo OpenAI | Dedicado por provedor com validação de modelo | -| `POST /v1/provedores/{provedor}/embeddings` | Incorporações OpenAI | Dedicado por provedor com validação de modelo | -| `POST /v1/provedores/{provedor}/imagens/gerações` | Imagens OpenAI | Dedicado por provedor com validação de modelo | -| `POST /v1/messages/count_tokens` | Contagem de tokens de Claude | Rota API | -| `OBTER /v1/modelos` | Lista de modelos OpenAI | Rota API (chat + incorporação + imagem + modelos customizados) | -| `GET /api/models/catalog` | Catálogo | Todos os modelos agrupados por fornecedor + tipo | -| `POST /v1beta/models/*:streamGenerateContent` | Nativo de Gêmeos | Rota API | -| `GET/PUT/DELETE /api/settings/proxy` | Configuração de proxy | Configuração de proxy de rede | -| `POST /api/settings/proxy/test` | Conectividade proxy | Endpoint de teste de integridade/conectividade do proxy | -| `GET/POST/DELETE /api/provider-models` | Modelos de Provedores | Metadados de modelo de provedor que respaldam modelos disponíveis personalizados e gerenciados |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -O manipulador de bypass (`open-sse/utils/bypassHandler.ts`) intercepta solicitações "descartáveis" conhecidas da Claude CLI — pings de aquecimento, extrações de títulos e contagens de tokens — e retorna uma**resposta falsa**sem consumir tokens do provedor upstream. Isso é acionado apenas quando `User-Agent` contém `claude-cli`.## Request Logger Pipeline +## Bypass Handler -O registrador de solicitações (`open-sse/utils/requestLogger.ts`) fornece um pipeline de registro de depuração de 7 estágios, desabilitado por padrão, habilitado via `ENABLE_REQUEST_LOGS=true`:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Os arquivos são gravados em `/logs//` para cada sessão de solicitação.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- resfriamento da conta do provedor em erros transitórios/taxa/autenticação -- fallback da conta antes da falha na solicitação -- modelo combinado substituto quando o caminho do modelo/provedor atual se esgota## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- pré-verificação e atualização com nova tentativa para provedores atualizáveis -- Nova tentativa 401/403 após tentativa de atualização no caminho principal## 3) Stream Safety +## 2) Token Expiry -- controlador de fluxo com reconhecimento de desconexão -- fluxo de tradução com liberação de fim de fluxo e tratamento `[DONE]` -- fallback de estimativa de uso quando faltam metadados de uso do provedor## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- erros de sincronização aparecem, mas o tempo de execução local continua -- o agendador tem lógica com capacidade de repetição, mas a execução periódica atualmente chama a sincronização de tentativa única por padrão## 5) Data Integrity +## 3) Stream Safety -- Migrações de esquema SQLite e ganchos de atualização automática na inicialização -- legado JSON → caminho de compatibilidade de migração SQLite## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Fontes de visibilidade em tempo de execução: +## 4) Cloud Sync Degradation -- logs do console de `src/sse/utils/logger.ts` -- agregados de uso por solicitação no SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- capturas detalhadas de carga útil em quatro estágios no SQLite (`request_detail_logs`) quando `settings.detailed_logs_enabled=true` -- registro de status de solicitação textual em `log.txt` (opcional/compatível) -- logs opcionais de solicitação/tradução profunda em `logs/` quando `ENABLE_REQUEST_LOGS=true` -- endpoints de uso do painel (`/api/usage/*`) para consumo da UI +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -A captura detalhada da carga útil da solicitação armazena até quatro estágios de carga útil JSON por chamada roteada: +## 5) Data Integrity -- solicitação bruta recebida do cliente -- solicitação traduzida realmente enviada upstream -- resposta do provedor reconstruída como JSON; as respostas transmitidas são compactadas no resumo final mais os metadados do fluxo -- resposta final do cliente retornada pelo OmniRoute; as respostas transmitidas são armazenadas no mesmo formulário de resumo compacto## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- O segredo JWT (`JWT_SECRET`) protege a verificação/assinatura de cookies da sessão do painel -- O bootstrap de senha inicial (`INITIAL_PASSWORD`) deve ser configurado explicitamente para provisionamento de primeira execução -- O segredo HMAC da chave de API (`API_KEY_SECRET`) protege o formato de chave de API local gerado -- Os segredos do provedor (chaves/tokens de API) persistem no banco de dados local e devem ser protegidos no nível do sistema de arquivos -- Os endpoints de sincronização em nuvem dependem da semântica de autenticação de chave de API + ID de máquina## Environment and Runtime Matrix +## Observability and Operational Signals -Variáveis de ambiente usadas ativamente pelo código: +Runtime visibility sources: -- Aplicativo/autenticação: `JWT_SECRET`, `INITIAL_PASSWORD` -- Armazenamento: `DATA_DIR` -- Comportamento do nó compatível: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Substituição opcional da base de armazenamento (Linux/macOS quando `DATA_DIR` não definido): `XDG_CONFIG_HOME` -- Hash de segurança: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Registro: `ENABLE_REQUEST_LOGS` -- URL de sincronização/nuvem: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- Proxy de saída: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` e variantes em letras minúsculas -- Sinalizadores de recurso SOCKS5: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- Auxiliares de plataforma/tempo de execução (não configuração específica do aplicativo): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` e `localDb` compartilham a mesma política de diretório base (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) com migração de arquivo legado. -2. `/api/v1/route.ts` delega para o mesmo construtor de catálogo unificado usado por `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) para evitar desvio semântico. -3. O registrador de solicitações grava cabeçalhos/corpo completos quando habilitado; trate o diretório de log como confidencial. -4. O comportamento da nuvem depende do `NEXT_PUBLIC_BASE_URL` correto e da acessibilidade do endpoint da nuvem. -5. O diretório `open-sse/` é publicado como `@omniroute/open-sse`**pacote de espaço de trabalho npm**. O código-fonte o importa via `@omniroute/open-sse/...` (resolvido por Next.js `transpilePackages`). Os caminhos de arquivo neste documento ainda usam o nome de diretório `open-sse/` para consistência. -6. Os gráficos no painel usam**Recharts**(baseados em SVG) para visualizações analíticas interativas e acessíveis (gráficos de barras de uso de modelo, tabelas de detalhamento de fornecedores com taxas de sucesso). -7. Os testes E2E usam**Playwright**(`tests/e2e/`), executados via `npm run test:e2e`. Os testes de unidade usam**Node.js test runner**(`tests/unit/`), executados via `npm run test:unit`. O código fonte em `src/` é**TypeScript**(`.ts`/`.tsx`); o espaço de trabalho `open-sse/` permanece JavaScript (`.js`). -8. A página de configurações é organizada em 5 guias: Segurança, Roteamento (6 estratégias globais: preenchimento primeiro, round-robin, p2c, aleatório, menos usado, com custo otimizado), Resiliência (limites de taxa editáveis, disjuntor, políticas), IA (pensando no orçamento, prompt do sistema, cache de prompt), Avançado (proxy).## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- Compilar a partir do código-fonte: `npm run build` -- Construa a imagem do Docker: `docker build -t omniroute .` -- Inicie o serviço e verifique: -- `GET /api/configurações` +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` - `GET /api/v1/models` -- O URL base de destino da CLI deve ser `http://:20128/v1` quando `PORT=20128` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/pt/docs/FEATURES.md b/docs/i18n/pt/docs/FEATURES.md index 35fa81f662..85e0bd2005 100644 --- a/docs/i18n/pt/docs/FEATURES.md +++ b/docs/i18n/pt/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Guia visual para cada seção do painel do OmniRoute.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Gerencie conexões de provedores de IA: provedores OAuth (Claude Code, Codex, Gemini CLI), provedores de chaves de API (Groq, DeepSeek, OpenRouter) e provedores gratuitos (Qoder, Qwen, Kiro). As contas Kiro incluem rastreamento de saldo de crédito – créditos restantes, subsídio total e data de renovação visíveis em Painel → Uso.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Crie combinações de modelos de roteamento com 6 estratégias: prioridade, ponderada, round-robin, aleatória, menos usada e com custo otimizado. Cada combinação encadeia vários modelos com fallback automático e inclui modelos rápidos e verificações de prontidão.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Análise de uso abrangente com consumo de tokens, estimativas de custos, mapas de calor de atividades, gráficos de distribuição semanais e detalhamentos por provedor.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Monitoramento em tempo real: tempo de atividade, memória, versão, percentis de latência (p50/p95/p99), estatísticas de cache e estados de disjuntores do provedor.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Quatro modos para depurar traduções de API:**Playground**(conversor de formato),**Chat Tester**(solicitações ao vivo),**Test Bench**(testes em lote) e**Live Monitor**(transmissão em tempo real).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Teste qualquer modelo diretamente do painel. Selecione provedor, modelo e endpoint, escreva prompts com o Monaco Editor, transmita respostas em tempo real, aborte o mid-stream e visualize métricas de tempo.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Temas de cores personalizáveis ​​para todo o painel. Escolha entre 7 cores predefinidas (Coral, Azul, Vermelho, Verde, Violeta, Laranja, Ciano) ou crie um tema personalizado escolhendo qualquer cor hexadecimal. Suporta modo claro, escuro e sistema.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Painel de configurações abrangente com guias: +Comprehensive settings panel with tabs: --**Geral**— Armazenamento do sistema, gerenciamento de backup (exportar/importar banco de dados) -**Aparência**— Seletor de tema (escuro/claro/sistema), predefinições de tema de cores e cores personalizadas, visibilidade do registro de saúde, controles de visibilidade de itens da barra lateral -**Segurança**— Proteção de endpoint de API, bloqueio de provedor personalizado, filtragem de IP, informações de sessão -**Roteamento**— Aliases de modelo, degradação de tarefas em segundo plano -**Resiliência**— Persistência de limite de taxa, ajuste de disjuntor, desativação automática de contas banidas, monitoramento de expiração de provedor -**Avançado**— Substituições de configuração, trilha de auditoria de configuração, modo de degradação de fallback![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Configuração com um clique para ferramentas de codificação de IA: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor e Factory Droid. Apresenta aplicação/redefinição de configuração automatizada, perfis de conexão e mapeamento de modelo.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Painel para descobrir e gerenciar agentes CLI. Mostra uma grade de 14 agentes integrados (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) com: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Status da instalação**— Instalado/Não encontrado com detecção de versão -**Selos de protocolo**— stdio, HTTP, etc. -**Agentes personalizados**— Registre qualquer ferramenta CLI via formulário (nome, binário, comando de versão, spawn args) -**CLI Fingerprint Matching**— Alternância por provedor para corresponder às assinaturas de solicitação CLI nativas, reduzindo o risco de banimento e preservando o IP do proxy--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Gere imagens, vídeos e músicas a partir do painel. Suporta OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open e MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Registro de solicitações em tempo real com filtragem por provedor, modelo, conta e chave de API. Mostra códigos de status, uso de token, latência e detalhes de resposta.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Seu endpoint de API unificado com detalhamento de recursos: conclusões de bate-papo, API de respostas, incorporações, geração de imagens, reclassificação, transcrição de áudio, conversão de texto em fala, moderações e chaves de API registradas. Integração do Cloudflare Quick Tunnel e suporte de proxy em nuvem para acesso remoto.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -Crie, escopo e revogue chaves de API. Cada chave pode ser restrita a modelos/provedores específicos com acesso total ou permissões somente leitura. Gerenciamento visual de chaves com rastreamento de uso.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Rastreamento de ações administrativas com filtragem por tipo de ação, ator, alvo, endereço IP e carimbo de data/hora. Histórico completo de eventos de segurança.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Aplicativo de desktop Native Electron para Windows, macOS e Linux. Execute o OmniRoute como um aplicativo independente com integração à bandeja do sistema, suporte offline, atualização automática e instalação com um clique. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Principais recursos: +Key features: -- Pesquisa de prontidão do servidor (sem tela em branco na inicialização a frio) -- Bandeja do sistema com gerenciamento de portas -- Política de Segurança de Conteúdo -- Bloqueio de instância única -- Atualização automática ao reiniciar -- UI condicional à plataforma (semáforos macOS, barra de título padrão do Windows/Linux) -- Pacote de compilação Hardened Electron — `node_modules` com link simbólico no pacote independente é detectado e rejeitado antes do empacotamento, evitando a dependência de tempo de execução na máquina de compilação (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Consulte [`electron/README.md`](../electron/README.md) para documentação completa. +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/pt/docs/TROUBLESHOOTING.md b/docs/i18n/pt/docs/TROUBLESHOOTING.md index d7e69f0999..382b3717f0 100644 --- a/docs/i18n/pt/docs/TROUBLESHOOTING.md +++ b/docs/i18n/pt/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -Problemas e soluções comuns para OmniRoute.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Problema | Solução | -| ----------------------------------------- | -------------------------------------------------------------------------------------- | --- | -| O primeiro login não funciona | Defina `INITIAL_PASSWORD` em `.env` (sem padrão codificado) | -| Painel abre na porta errada | Defina `PORT=20128` e `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Nenhum registro de solicitação em `logs/` | Definir `ENABLE_REQUEST_LOGS=true` | -| EACCES: permissão negada | Defina `DATA_DIR=/path/to/writable/dir` para substituir `~/.omniroute` | -| Estratégia de roteamento não salva | Atualização para v1.4.11+ (correção do esquema Zod para persistência de configurações) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Causa:**Cota do provedor esgotada. +**Cause:** Provider quota exhausted. -**Correção:** +**Fix:** -1. Verifique o rastreador de cota do painel -2. Use um combo com níveis alternativos -3. Mude para um nível mais barato/gratuito### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Causa:**Cota de assinatura esgotada. +### Rate Limiting -**Correção:** +**Cause:** Subscription quota exhausted. -- Adicionar substituto: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Use GLM/MiniMax como backup barato### OAuth Token Expired +**Fix:** -OmniRoute atualiza automaticamente os tokens. Se os problemas persistirem: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Painel → Provedor → Reconectar -2. Exclua e adicione novamente a conexão do provedor--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Verifique se `BASE_URL` aponta para sua instância em execução (por exemplo, `http://localhost:20128`) -2. Verifique os pontos `CLOUD_URL` para o seu endpoint de nuvem (por exemplo, `https://omniroute.dev`) -3. Mantenha os valores `NEXT_PUBLIC_*` alinhados com os valores do lado do servidor### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Sintoma:**`Token inesperado 'd'...` no endpoint da nuvem para chamadas sem streaming. +### Cloud `stream=false` Returns 500 -**Causa:**O upstream retorna a carga SSE enquanto o cliente espera JSON. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Solução alternativa:**use `stream=true` para chamadas diretas na nuvem. O tempo de execução local inclui substituto SSE→JSON.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Crie uma nova chave no painel local (`/api/keys`) -2. Execute a sincronização na nuvem: Habilite Nuvem → Sincronizar agora -3. Chaves antigas/não sincronizadas ainda podem retornar `401` na nuvem--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Verifique os campos de tempo de execução: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. Para modo portátil: use o destino de imagem `runner-cli` (CLIs agrupados) -3. Para o modo de montagem do host: defina `CLI_EXTRA_PATHS` e monte o diretório bin do host como somente leitura -4. Se `installed=true` e `runnable=false`: o binário foi encontrado, mas falhou na verificação de integridade### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Verifique as estatísticas de uso em Painel → Uso -2. Mude o modelo primário para GLM/MiniMax -3. Use o nível gratuito (Gemini CLI, Qoder) para tarefas não críticas -4. Defina orçamentos de custos por chave de API: Painel → Chaves de API → Orçamento--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Defina `ENABLE_REQUEST_LOGS=true` em seu arquivo `.env`. Os logs aparecem no diretório `logs/`.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Estado principal: `${DATA_DIR}/storage.sqlite` (provedores, combos, aliases, chaves, configurações) -- Uso: tabelas SQLite em `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + opcionais `${DATA_DIR}/log.txt` e `${DATA_DIR}/call_logs/` -- Solicitar logs: `/logs/...` (quando `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Quando o disjuntor de um provedor está ABERTO, as solicitações são bloqueadas até que o tempo de espera expire. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Correção:** +**Fix:** -1. Vá para**Painel → Configurações → Resiliência** -2. Verifique a placa do disjuntor do provedor afetado -3. Clique em**Redefinir tudo**para limpar todos os disjuntores ou aguarde o tempo de espera expirar -4. Verifique se o provedor está realmente disponível antes de redefinir### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Se um provedor entrar repetidamente no estado OPEN: +### Provider keeps tripping the circuit breaker -1. Verifique**Dashboard → Health → Provider Health**para ver o padrão de falha -2. Vá para**Configurações → Resiliência → Perfis do Provedor**e aumente o limite de falha -3. Verifique se o provedor alterou os limites da API ou requer nova autenticação -4. Revise a telemetria de latência – alta latência pode causar falhas baseadas em tempo limite--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Certifique-se de usar o prefixo correto: `deepgram/nova-3` ou `assemblyai/best` -- Verifique se o provedor está conectado em**Painel → Provedores**### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Verifique os formatos de áudio suportados: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- Verifique se o tamanho do arquivo está dentro dos limites do provedor (normalmente <25 MB) -- Verifique a validade da chave API do provedor no cartão do provedor--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Use**Dashboard → Tradutor**para depurar problemas de tradução de formato: +Use **Dashboard → Translator** to debug format translation issues: -| Modo | Quando usar | -| ------------------------- | --------------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Parque Infantil** | Compare os formatos de entrada/saída lado a lado — cole uma solicitação com falha para ver como ela é traduzida | -| **Testador de bate-papo** | Envie mensagens ao vivo e inspecione a carga completa de solicitação/resposta, incluindo cabeçalhos | -| **Banco de testes** | Execute testes em lote em combinações de formatos para descobrir quais traduções estão quebradas | -| **Monitoramento ao vivo** | Observe o fluxo de solicitações em tempo real para detectar problemas intermitentes de tradução | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Tags de pensamento não aparecem**— Verifique se o provedor alvo apoia o pensamento e a configuração do orçamento de pensamento -**Queda de chamadas de ferramentas**— Algumas traduções de formato podem remover campos não suportados; verificar no modo Playground -**Prompt do sistema ausente**— Claude e Gemini lidam com os prompts do sistema de maneira diferente; verifique o resultado da tradução -**SDK retorna string bruta em vez de objeto**— Corrigido na v1.1.0: o Response Sanitizer agora remove campos não padrão (`x_groq`, `usage_breakdown`, etc.) que causam falhas de validação do OpenAI SDK Pydantic -**GLM/ERNIE rejeita função `sistema`**— Corrigido na v1.1.0: o normalizador de função mescla automaticamente mensagens do sistema em mensagens do usuário para modelos incompatíveis -**função `developer` não reconhecida**— Corrigido na v1.1.0: convertido automaticamente para `system` para provedores não-OpenAI -**`json_schema` não funciona com Gemini**— Corrigido na v1.1.0: `response_format` agora é convertido para `responseMimeType` + `responseSchema` do Gemini--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- O limite automático de taxa se aplica apenas a provedores de chaves de API (não a OAuth/assinatura) -- Verifique se**Configurações → Resiliência → Perfis do Provedor**tem limite de taxa automática ativado -- Verifique se o provedor retorna códigos de status `429` ou cabeçalhos `Retry-After`### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -Os perfis do provedor oferecem suporte a estas configurações: +### Tuning exponential backoff --**Atraso base**— Tempo de espera inicial após a primeira falha (padrão: 1s) -**Atraso máximo**— Limite máximo de tempo de espera (padrão: 30s) -**Multiplicador**— Quanto aumentar o atraso por falha consecutiva (padrão: 2x)### Anti-thundering herd +Provider profiles support these settings: -Quando muitas solicitações simultâneas atingem um provedor com taxa limitada, o OmniRoute usa mutex + limitação automática de taxa para serializar solicitações e evitar falhas em cascata. Isso é automático para provedores de chaves de API.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Alguns usuários do OmniRoute colocam o gateway na frente do RAG ou das pilhas de agentes. Nessas configurações é comum ver um padrão estranho: OmniRoute parece íntegro (provedores ativos, perfis de roteamento ok, sem alertas de limite de taxa), mas a resposta final ainda está errada. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -Na prática, esses incidentes geralmente vêm do pipeline RAG downstream e não do gateway em si. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Se você deseja um vocabulário compartilhado para descrever essas falhas, você pode usar o WFGY ProblemMap, um recurso de texto de licença externa do MIT que define dezesseis padrões recorrentes de falha RAG/LLM. Em alto nível, abrange: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- desvio de recuperação e limites de contexto quebrados -- índices vazios ou obsoletos e armazenamentos de vetores -- incorporação versus incompatibilidade semântica -- problemas de montagem imediata e janela de contexto -- colapso lógico e respostas excessivamente confiantes -- falhas de coordenação de cadeia longa e de agente -- memória multiagente e desvio de função -- problemas de implantação e ordenação de bootstrap +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -A ideia é simples: +The idea is simple: -1. Ao investigar uma resposta incorreta, capture: - - tarefa e solicitação do usuário - - combinação de rota ou provedor no OmniRoute - - qualquer contexto RAG usado posteriormente (documentos recuperados, chamadas de ferramentas, etc.) -2. Mapeie o incidente para um ou dois números do ProblemMap do WFGY (`No.1` … `No.16`). -3. Armazene o número em seu próprio painel, runbook ou rastreador de incidentes próximo aos logs do OmniRoute. -4. Use a página WFGY correspondente para decidir se você precisa alterar sua pilha RAG, recuperador ou estratégia de roteamento. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -O texto completo e as receitas concretas estão aqui (licença MIT, somente texto): +Full text and concrete recipes live here (MIT license, text only): -[README do WFGY ProblemMap](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) +[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -Você pode ignorar esta seção se não executar RAG ou pipelines de agente atrás do OmniRoute.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**Problemas do GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Arquitetura**: Consulte [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) para obter detalhes internos -**Referência de API**: Consulte [`docs/API_REFERENCE.md`](API_REFERENCE.md) para todos os endpoints -**Painel de saúde**: verifique**Painel → Saúde**para ver o status do sistema em tempo real -**Tradutor**: Use**Dashboard → Tradutor**para depurar problemas de formato +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/pt/llm.txt b/docs/i18n/pt/llm.txt new file mode 100644 index 0000000000..0b3f167424 --- /dev/null +++ b/docs/i18n/pt/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Português (Portugal)) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Visão Geral + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Segurança +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/ro/README.md b/docs/i18n/ro/README.md index e994004e90..fe29812f84 100644 --- a/docs/i18n/ro/README.md +++ b/docs/i18n/ro/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Proxy-ul dvs. universal API — un punct final, peste 60 de furnizori, zero timpi de nefuncționare. Acum cu**Server MCP (25 de instrumente)**,**Protocol A2A**,**Memory/Skills Systems**și**Electron Desktop App**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Finalizări de chat • Încorporare • Generare de imagini • Video • Muzică • Audio • Reclasificare •**Căutare Web**• Server MCP • Protocol A2A • 100% TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Proxy-ul dvs. universal API — un punct final, peste 60 de furnizori, zero tim [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Site web](https://omniroute.online) • [🚀 Pornire rapidă](#-pornire rapidă) • [💡 Funcții](#-funcții-cheie) • [📖 Documente](#-documentație) • [💰 Preț](#-preț-din-o privire) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Disponibil în:**🇺🇸 [Engleză](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugalia)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filippine](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,28 +60,30 @@ _Proxy-ul dvs. universal API — un punct final, peste 60 de furnizori, zero tim ## 📸 Dashboard Preview - -Faceți clic pentru a vedea capturi de ecran din tabloul de bord +
+Click to see dashboard screenshots -| Pagina | Captură de ecran | -| ------------------------ | ----------------------------------------------------- | ---------- | -| **Furnizori** | ![Furnizori](docs/screenshots/01-providers.png) | -| **Combo** | ![Combo](docs/screenshots/02-combos.png) | -| **Analitice** | ![Analytics](docs/screenshots/03-analytics.png) | -| **Sănătate** | ![Health](docs/screenshots/04-health.png) | -| **Translator** | ![Translator](docs/screenshots/05-translator.png) | -| **Setări** | ![Setări](docs/screenshots/06-settings.png) | -| **Instrumente CLI** | ![Instrumente CLI](docs/screenshots/07-cli-tools.png) | -| **Jurnale de utilizare** | ![Utilizare](docs/screenshots/08-usage.png) | -| **Punctele finale** | ![Endpoints](docs/screenshots/09-endpoint.png) |
| +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | + + --- ### 🤖 Free AI Provider for your favorite coding agents -_Conectați orice instrument IDE sau CLI alimentat de AI prin OmniRoute — gateway API gratuit pentru codare nelimitată._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ - + @@ -125,481 +134,555 @@ _Conectați orice instrument IDE sau CLI alimentat de AI prin OmniRoute — gate Codex CLI
Codex CLI
- ⭐ 60,8K + ⭐ 60.8K - +
@@ -88,28 +97,28 @@ _Conectați orice instrument IDE sau CLI alimentat de AI prin OmniRoute — gate NanoBot
NanoBot

- ⭐ 20,9K + ⭐ 20.9K
PicoClaw
PicoClaw

- ⭐ 14,6K + ⭐ 14.6K
ZeroClaw
ZeroClaw

- ⭐ 9,9K + ⭐ 9.9K
IronClaw
- Gheară de Fier + IronClaw

- ⭐ 2,1K + ⭐ 2.1K
Claude Code
Claude Code

- ⭐ 67,3K + ⭐ 67.3K
Gemini CLI
- CLI Gemini + Gemini CLI

- ⭐ 94,7K + ⭐ 94.7K
Kilo Code
- Cod kilo + Kilo Code

- ⭐ 15,5K + ⭐ 15.5K
-📡 Toți agenții se conectează prin http://localhost:20128/v1 sau http://cloud.omniroute.online/v1 — o singură configurare, modele și cotă nelimitate--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Nu mai risipi banii și nu mai atingeți limitele:** +**Stop wasting money and hitting limits:** -- Cota de abonament expiră neutilizată în fiecare lună -- Limitele ratelor vă opresc codificarea la mijloc -- API-uri scumpe (20-50 USD/lună per furnizor) -- Comutare manuală între furnizori +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute rezolvă asta:** +**OmniRoute solves this:** -- ✅**Maximizați abonamentele**- Urmăriți cota, utilizați fiecare bit înainte de resetare -- ✅**Auto de rezervă**- Abonament → Cheie API → Ieftin → Gratuit, timp de nefuncționare zero -- ✅**Multi-cont**- Round-robin între conturi pentru fiecare furnizor -- ✅**Universal**- Funcționează cu Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, orice instrument CLI--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Alăturați-vă comunității noastre!**[WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Obțineți ajutor, împărtășiți sfaturi și rămâneți la curent. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Site web**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Probleme**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Grup comunitar](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Contribuind**: Consultați [CONTRIBUTING.md](CONTRIBUTING.md), deschideți un PR sau alegeți o „prima ediție bună” -**Proiect original**: [9router by decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Când deschideți o problemă, rulați comanda system-info și atașați fișierul generat:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Acest lucru generează un `system-info.txt` cu versiunea dvs. Node.js, versiunea OmniRoute, detaliile sistemului de operare, instrumentele CLI instalate (qoder, gemini, claude, codex, antigravity, droid etc.), starea Docker/PM2 și pachetele de sistem - tot ce avem nevoie pentru a reproduce problema rapid. Atașați fișierul direct la problema dvs. GitHub.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Fiecare dezvoltator care folosește instrumente AI se confruntă zilnic cu aceste probleme.**OmniRoute a fost creat pentru a le rezolva pe toate - de la depășiri de costuri la blocaje regionale, de la fluxuri OAuth întrerupte la operațiuni de protocol și observabilitate a întreprinderii. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. - -💸 1. „Plătesc pentru un abonament scump, dar tot sunt întrerupt de limite” +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Dezvoltatorii plătesc 20–200 USD/lună pentru Claude Pro, Codex Pro sau GitHub Copilot. Chiar și plătind, cota are un plafon - 5 ore de utilizare, limite săptămânale sau limite de tarif pe minut. La mijlocul sesiunii de codare, furnizorul nu mai răspunde și dezvoltatorul își pierde fluxul și productivitatea. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Cum o rezolvă OmniRoute:** +**How OmniRoute solves it:** --**Smart 4-Tier Fallback**— Dacă cota de abonament se epuizează, redirecționează automat la cheia API → Ieftin → Gratuit fără intervenție manuală --**Urmărirea limitelor furnizorului**— Instantaneele de cotă stocate în cache se reîmprospătează pe o programare pe partea de server (implicit `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) cu reîmprospătare manuală disponibilă în interfața de utilizare --**Asistență pentru mai multe conturi**— Conturi multiple per furnizor cu turneu automat automat — când unul se epuizează, trece la următorul --**Combinații personalizate**— Lanțuri de rezervă personalizabile cu 9 strategii de echilibrare (prioritar, ponderat, completare mai întâi, round-robin, P2C, aleatoriu, cel mai puțin utilizat, optimizat din punct de vedere al costurilor, strict aleatoriu) --**Cote de afaceri Codex**— Monitorizarea cotelor de spațiu de lucru pentru afaceri/echipe direct în tabloul de bord
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard - -🔌 2. „Trebuie să folosesc mai mulți furnizori, dar fiecare are un API diferit” + -OpenAI folosește un format, Claude (Anthropic) folosește altul, Gemini încă altul. Dacă un dezvoltator dorește să testeze modele de la diferiți furnizori sau să se retragă între aceștia, trebuie să reconfigureze SDK-urile, să schimbe punctele finale, să se ocupe de formate incompatibile. Furnizorii personalizați (FriendLI, NIM) au puncte finale de model non-standard. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Cum o rezolvă OmniRoute:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Unified Endpoint**— Un singur `http://localhost:20128/v1` servește ca proxy pentru toți cei peste 60 de furnizori --**Traducerea formatului**— Automată și transparentă: OpenAI ↔ Claude ↔ Gemeni ↔ Responses API --**Response Sanitization**— Elimina câmpurile non-standard (`x_groq`, `usage_breakdown`, `service_tier`) care încalcă OpenAI SDK v1.83+ --**Normalizarea rolurilor**— Convertește `dezvoltator` → `sistem` pentru furnizorii non-OpenAI; `sistem` → `utilizator` pentru GLM/ERNIE --**Think Tag Extraction**— Extrage blocurile `` din modele precum DeepSeek R1 în `conținut_raționament` standardizat --**Ieșire structurată pentru Gemini**— `json_schema` → `responseMimeType`/`responseSchema` conversie automată --**`stream` este implicit `false`**— Se aliniază cu specificațiile OpenAI, evitând SSE neașteptat în SDK-urile Python/Rust/Go
+**How OmniRoute solves it:** - -🌐 3. „Furnizorul meu de AI îmi blochează regiunea/țara” +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Furnizori precum OpenAI/Codex blochează accesul din anumite regiuni geografice. Utilizatorii primesc erori precum „unsupported_country_region_territory” în timpul conexiunilor OAuth și API. Acest lucru este frustrant în special pentru dezvoltatorii din țările în curs de dezvoltare. + -**Cum o rezolvă OmniRoute:** +
+🌐 3. "My AI provider blocks my region/country" --**3-Level Proxy Config**— Proxy configurabil la 3 niveluri: global (tot traficul), per furnizor (doar un singur furnizor) și per conexiune/cheie --**Insigne de proxy cu coduri de culoare**— Indicatori vizuali: 🟢 proxy global, 🟡 proxy furnizor, 🔵 proxy de conexiune, indicând întotdeauna IP-ul --**Schimb de jetoane OAuth prin proxy**— fluxul OAuth trece și prin proxy, rezolvând `unsupported_country_region_territory` --**Teste de conexiune prin proxy**— Testele de conexiune folosesc proxy-ul configurat (nu mai este ocolire directă) --**Support SOCKS5**— Suport complet SOCKS5 proxy pentru rutarea de ieșire --**TLS Fingerprint Spoofing**- Amprenta TLS asemănătoare unui browser prin `wreq-js` pentru a ocoli detectarea botului --**🔏 Potrivirea amprentei CLI**— Reordonează anteturile și câmpurile de corp pentru a se potrivi cu semnăturile binare CLI native, reducând drastic riscul de semnalare a contului. IP-ul proxy este păstrat - obțineți simultan mascarea IP stealth**și**
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. - -🆓 4. „Vreau să folosesc AI pentru codare, dar nu am bani” +**How OmniRoute solves it:** -Nu toată lumea poate plăti 20–200 USD/lună pentru abonamentele AI. Studenții, dezvoltatorii din țările emergente, pasionații și freelancerii au nevoie de acces la modele de calitate la cost zero. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Cum o rezolvă OmniRoute:** + --**Free Tier Providers Built-in**— Suport nativ pentru furnizori 100% gratuiti: Qoder (5 modele nelimitate prin OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 modele nelimitate: q-coder-3-fwen: qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID gratuit), Gemini CLI (180.000 de jetoane/lună gratuit) --**Ollama Cloud**— Modele Ollama găzduite în cloud la `api.ollama.com` cu nivelul gratuit „Utilizare ușoară”; utilizați prefixul `ollamacloud/` --**Free-Only Combo**— Lanț `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = 0 USD/lună fără timp de nefuncționare --**NVIDIA NIM Free Access**— ~40 RPM dev-forever acces gratuit la peste 70 de modele la build.nvidia.com (tranziție de la credite la limitele de rate pur) --**Cost Optimized Strategy**— Strategie de rutare care alege automat cel mai ieftin furnizor disponibil +
+🆓 4. "I want to use AI for coding but I have no money" - -🔒 5. „Trebuie să-mi protejez poarta AI de accesul neautorizat” +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -Când expuneți un gateway AI în rețea (LAN, VPS, Docker), oricine are adresa poate consuma jetoanele/cota dezvoltatorului. Fără protecție, API-urile sunt vulnerabile la utilizare greșită, injectare promptă și abuz. +**How OmniRoute solves it:** -**Cum o rezolvă OmniRoute:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**Gestionarea cheilor API**— Generare, rotație și stabilire a domeniului pentru fiecare furnizor cu o pagină dedicată „/dashboard/api-manager” --**Permisiuni la nivel de model**— Restricționați cheile API la anumite modele (`openai/*`, modele de metacaractere), cu comutarea Permite tot/Restricționați --**API Endpoint Protection**— Solicitați o cheie pentru `/v1/models` și blocați anumiți furnizori din listă --**Auth Guard + CSRF Protection**— Toate rutele tabloului de bord sunt protejate cu middleware `withAuth` + jetoane CSRF --**Rate Limiter**— Limitarea ratei per-IP cu ferestre configurabile --**Filtrare IP**— Lista permisă/lista blocată pentru controlul accesului --**Prompt Injection Guard**— Igienizare împotriva tiparelor de prompte rău intenționate --**Criptare AES-256-GCM**— Acreditări criptate în repaus
+ - -🛑 6. „Furnizorul meu a căzut și mi-am pierdut fluxul de codare” +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -Furnizorii de AI pot deveni instabili, pot returna erori 5xx sau pot atinge limitele temporare ale ratei. Dacă un dezvoltator depinde de un singur furnizor, acesta este întrerupt. Fără întreruptoare, reîncercări repetate pot bloca aplicația. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Cum o rezolvă OmniRoute:** +**How OmniRoute solves it:** --**Circuit Breaker per-model**- Deschidere/închidere automată cu praguri configurabile și răcire (Închis/Deschis/Pe jumătate deschis), pentru fiecare model pentru a evita blocurile în cascadă --**Backoff exponențial**— Întârzieri progresive ale reîncercării --**Anti-Thundering Herd**— Mutex + protecție semafor împotriva furtunilor concurente de reîncercare --**Combo Fallback Chains**— Dacă furnizorul principal eșuează, trece automat prin lanț fără nicio intervenție --**Combo Circuit Breaker**— Dezactivează automat furnizorii care eșuează dintr-un lanț combinat --**Tabloul de bord pentru sănătate**— Monitorizare timp de funcționare, stări întrerupătoare de circuit, blocări, statistici cache, latență p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest - -🔧 7. „Configurarea fiecărui instrument AI este plictisitoare și repetitivă” + -Dezvoltatorii folosesc Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Fiecare instrument are nevoie de o configurație diferită (punct final API, cheie, model). Reconfigurarea la schimbarea de furnizor sau de model este o pierdere de timp. +
+🛑 6. "My provider went down and I lost my coding flow" -**Cum o rezolvă OmniRoute:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**CLI Tools Dashboard**— pagină dedicată cu setare cu un singur clic pentru Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline --**GitHub Copilot Config Generator**— Generează `chatLanguageModels.json` pentru VS Code cu selecție de model în bloc --**Onboarding Wizard**— Configurare ghidată în 4 pași pentru utilizatorii debutanți --**Un punct final, toate modelele**— Configurați `http://localhost:20128/v1` o dată, accesați peste 60 de furnizori
+**How OmniRoute solves it:** - -🔑 8. „Gestionarea jetoanelor OAuth de la mai mulți furnizori este un iad” +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot - toate folosesc OAuth 2.0 cu token-uri care expiră. Dezvoltatorii trebuie să se reautentifice în mod constant, să se ocupe de `client_secret is missing`, `redirect_uri_mismatch` și eșecurile pe serverele de la distanță. OAuth pe LAN/VPS este deosebit de problematică. + -**Cum o rezolvă OmniRoute:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Reîmprospătare automată a simbolurilor**— jetoanele OAuth se reîmprospătează în fundal înainte de expirare --**OAuth 2.0 (PKCE) încorporat**— Flux automat pentru Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder --**OAuth cu mai multe conturi**— Conturi multiple per furnizor prin extragerea jetonului JWT/ID --**OAuth LAN/Remote Fix**— Detectare IP privată pentru `redirect_uri` + modul URL manual pentru servere la distanță --**OAuth în spatele Nginx**— Utilizează `window.location.origin` pentru compatibilitatea cu proxy invers --**Ghid OAuth la distanță**— Ghid pas cu pas pentru acreditările Google Cloud pe VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. - -📊 9. „Nu știu cât cheltuiesc sau unde” +**How OmniRoute solves it:** -Dezvoltatorii folosesc mai mulți furnizori plătiți, dar nu au o viziune unificată asupra cheltuielilor. Fiecare furnizor are propriul tablou de bord de facturare, dar nu există o vizualizare consolidată. Costurile neașteptate se pot acumula. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Cum o rezolvă OmniRoute:** + --**Tabloul de bord pentru analiza costurilor**— Urmărirea costurilor pe token și gestionarea bugetului per furnizor --**Limite bugetare pe nivel**— Plafonul de cheltuieli pe nivel care declanșează o rezervă automată --**Configurație de preț pe model**— Prețuri configurabile pe model --**Statistici de utilizare per cheie API**— Numărul de solicitări și marcajul temporal al ultimei utilizări per cheie --**Tabloul de bord de analiză**— Carduri cu statistici, diagramă de utilizare a modelului, tabel cu furnizori cu rate de succes și latență +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" - -🐛 10. „Nu pot diagnostica erorile și problemele în apelurile AI” +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Când un apel eșuează, dezvoltatorul nu știe dacă a fost o limită de rată, un simbol expirat, un format greșit sau o eroare a furnizorului. Jurnalele fragmentate pe diferite terminale. Fără observabilitate, depanarea este încercare și eroare. +**How OmniRoute solves it:** -**Cum o rezolvă OmniRoute:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Tabloul de bord pentru jurnalele unificate**— 4 file: jurnalele de solicitare, jurnalele proxy, jurnalele de audit, consolă --**Console Log Viewer**— Vizualizator în timp real în stil terminal cu niveluri codificate în culori, defilare automată, căutare, filtru --**SQLite Proxy Logs**— Jurnale persistente care supraviețuiesc repornirilor serverului --**Translator Playground**— 4 moduri de depanare: Playground (traducere format), Chat Tester (dus-întors), Test Bench (lot), Live Monitor (în timp real) --**Solicitare telemetrie**— latență p50/p95/p99 + urmărire X-Request-Id --**Înregistrare bazată pe fișiere cu rotație**— Jurnalele aplicațiilor se rotesc în funcție de dimensiune, zile de păstrare și număr de arhive; artefactele jurnalului de apeluri se rotesc în funcție de zilele de reținere și de numărul de fișiere --**System Info Report**— `npm run system-info` generează `system-info.txt` cu mediul dumneavoastră complet (versiunea Node, versiunea OmniRoute, OS, instrumente CLI, starea Docker/PM2). Atașați-l când raportați probleme pentru triaj instantaneu.
+ - -🏗️ 11. „Implementarea și întreținerea gateway-ului este complexă” +
+📊 9. "I don't know how much I'm spending or where" -Instalarea, configurarea și menținerea unui proxy AI în diferite medii (local, VPS, Docker, cloud) necesită multă muncă. Probleme precum căile hardcoded, `EACCES` pe directoare, conflictele de porturi și build-urile pe mai multe platforme adaugă fricțiuni. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Cum o rezolvă OmniRoute:** +**How OmniRoute solves it:** --**npm global install**— `npm install -g omniroute && omniroute` — gata --**Docker Multi-Platform**- AMD64 + ARM64 nativ (Apple Silicon, AWS Graviton, Raspberry Pi) --**Docker Compose Profiles**— `base` (fără instrumente CLI) și `cli` (cu Claude Code, Codex, OpenClaw) --**Electron Desktop App**— aplicație nativă pentru Windows/macOS/Linux cu bară de sistem, pornire automată, mod offline --**Split-Port Mode**— API și tablou de bord pe porturi separate pentru scenarii avansate (reverse proxy, rețea container) --**Cloud Sync**— Configurați sincronizarea între dispozitive prin Cloudflare Workers --**Backups DB**— Backup automat, restaurare, export și import al tuturor setărilor, cu `DISABLE_SQLITE_AUTO_BACKUP` pentru copiile de rezervă gestionate extern
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency - -🌍 12. „Interfața este doar în limba engleză și echipa mea nu vorbește engleză” + -Echipele din țările care nu vorbesc engleza, în special din America Latină, Asia și Europa, se luptă cu interfețele doar în limba engleză. Barierele lingvistice reduc adoptarea și cresc erorile de configurare. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Cum o rezolvă OmniRoute:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Tabloul de bord i18n — 30 de limbi**— Toate cele peste 500 de taste traduse, inclusiv arabă, bulgară, daneză, germană, spaniolă, finlandeză, franceză, ebraică, hindi, maghiară, indoneziană, italiană, japoneză, coreeană, malay, olandeză, norvegiană, poloneză, portugheză (PT/BR), română, rusă, slovacă, suedeză, thailandeză, ucraineană, filipineză, engleză, chineză, vietnameză, --**Suport RTL**— Suport de la dreapta la stânga pentru arabă și ebraică --**ReadME-uri în mai multe limbi**— 30 de traduceri complete de documentație --**Selector de limbă**— Pictograma glob în antet pentru comutare în timp real
+**How OmniRoute solves it:** - -🔄 13. „Am nevoie de mai mult decât de chat — am nevoie de încorporare, imagini, audio” +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -AI nu este doar finalizarea chatului. Dezvoltatorii trebuie să genereze imagini, să transcrie sunetul, să creeze înglobări pentru RAG, să reclasifice documentele și să modereze conținutul. Fiecare API are un punct final și un format diferit. + -**Cum o rezolvă OmniRoute:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Embeddings**— `/v1/embeddings` cu 6 furnizori și peste 9 modele --**Image Generation**— `/v1/images/generations` cu 10 furnizori și peste 20 de modele (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Text-to-Video**— `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) și SD WebUI --**Text-to-Music**— `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) --**Transcriere audio**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Text-to-Speech**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + furnizori existenți --**Moderații**— `/v1/moderations` — Verificări de siguranță a conținutului --**Reclasificare**— `/v1/rerank` — Reclasificarea relevanței documentului --**Responses API**— Suport complet `/v1/responses` pentru Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. - -🧪 14. „Nu am cum să testez și să compar calitatea între modele” +**How OmniRoute solves it:** -Dezvoltatorii vor să știe care model este cel mai bun pentru cazul lor de utilizare - cod, traducere, raționament - dar compararea manuală este lentă. Nu există instrumente de evaluare integrate. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**Cum o rezolvă OmniRoute:** + --**Evaluări LLM**— Testarea setului de aur cu 10 cazuri preîncărcate care acoperă salutări, matematică, geografie, generare de cod, conformitate cu JSON, traducere, reducere, refuz de siguranță --**4 strategii de potrivire**— `exact`, `contains`, `regex`, `custom` (funcția JS) --**Translator Playground Test Bench**— Testare în loturi cu mai multe intrări și rezultate așteptate, comparație între furnizori --**Tester de chat**— Tur complet dus-întors cu randare vizuală a răspunsului --**Live Monitor**— Flux în timp real al tuturor solicitărilor care circulă prin proxy +
+🌍 12. "The interface is English-only and my team doesn't speak English" - -📈 15. „Trebuie să mă extind fără a pierde performanța” +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -Pe măsură ce volumul cererilor crește, fără memorarea în cache aceleași întrebări generează costuri duplicate. Fara idempotenta, cererile duplicate procesarea deseurilor. Limitele de tarife pentru fiecare furnizor trebuie respectate. +**How OmniRoute solves it:** -**Cum o rezolvă OmniRoute:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Cache semantic**— Cache-ul pe două niveluri (semnătură + semantică) reduce costurile și latența --**Request Idempotency**— fereastră de deduplicare 5s pentru cereri identice --**Rate Limit Detection**— RPM per furnizor, interval minim și urmărire simultană maximă --**Limite de rată editabile**— Valori implicite configurabile în Setări → Reziliență cu persistență --**API Key Validation Cache**— cache pe 3 niveluri pentru performanța producției --**Tabloul de bord pentru sănătate cu telemetrie**— latență p50/p95/p99, statistici cache, timp de funcționare
+ - -🤖 16. „Vreau să controlez comportamentul modelului la nivel global” +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Dezvoltatori care doresc toate răspunsurile într-o anumită limbă, cu un anumit ton sau care doresc să limiteze simbolurile de raționament. Configurarea acestui lucru în fiecare instrument/cerere nu este practică. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Cum o rezolvă OmniRoute:** +**How OmniRoute solves it:** --**System Prompt Injection**— Prompt global aplicat tuturor solicitărilor --**Thinking Budget Validation**— Controlul raționării alocării token-ului per cerere (transmis, automat, personalizat, adaptiv) --**9 Strategii de rutare**— Strategii globale care determină modul în care sunt distribuite cererile --**Wildcard Router**— modelele `furnizor/*` direcționează dinamic către orice furnizor --**Combo Activare/Dezactivare Comutare**— Comută combo direct din tabloul de bord --**Comutare furnizor**— Activați/dezactivați toate conexiunile pentru un furnizor cu un singur clic --**Furnizori blocați**— Excludeți anumiți furnizori din lista `/v1/models`
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex - -🧰 17. „Am nevoie de instrumente MCP ca capabilități de produs de primă clasă” + -Multe gateway-uri AI expun MCP doar ca un detaliu ascuns de implementare. Echipele au nevoie de un nivel de operare vizibil și ușor de gestionat. +
+🧪 14. "I have no way to test and compare quality across models" -**Cum o rezolvă OmniRoute:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP apare în panoul de bord de navigare și fila de protocol final -- Pagina de management MCP dedicată cu proces, instrumente, domenii și audit -- Pornire rapidă încorporată pentru `omniroute --mcp` și integrarea clientului
+**How OmniRoute solves it:** - -🧠 18. „Am nevoie de orchestrare A2A cu sincronizare + căi de activități de flux” +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Fluxurile de lucru ale agenților necesită atât răspunsuri directe, cât și execuție în flux de lungă durată, cu control ciclului de viață. + -**Cum o rezolvă OmniRoute:** +
+📈 15. "I need to scale without losing performance" -- A2A JSON-RPC endpoint (`POST /a2a`) cu `message/send` și `message/stream` -- Streaming SSE cu propagare a stării terminale -- API-uri pentru ciclul de viață al sarcinilor pentru `tasks/get` și `tasks/cancel`
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. - -🛰️ 19. „Am nevoie de sănătate reală a procesului MCP, nu de stare ghicită” +**How OmniRoute solves it:** -Echipele operaționale trebuie să știe dacă MCP este de fapt în viață, nu doar dacă un API este accesibil. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**Cum o rezolvă OmniRoute:** + -- Fișier runtime heartbeat cu PID, marcaje de timp, transport, număr de instrumente și modul de aplicare -- API de stare MCP care combină bătăile inimii + activitatea recentă -- Carduri de stare a interfeței de utilizare pentru prospețimea procesului/uptime/inima +
+🤖 16. "I want to control model behavior globally" - -📋 20. „Am nevoie de o execuție auditabilă a instrumentului MCP” +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Când instrumentele modifică configurația sau declanșează acțiuni operaționale, echipele au nevoie de trasabilitate criminalistică. +**How OmniRoute solves it:** -**Cum o rezolvă OmniRoute:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- Înregistrare de audit susținută de SQLite pentru apelurile instrumentelor MCP -- Filtrează după instrument, succes/eșec, cheie API și paginare -- Tabelul de audit al tabloului de bord + punctele finale de statistici pentru automatizare
+ - -🔐 21. „Am nevoie de permisiuni MCP pentru fiecare integrare” +
+🧰 17. "I need MCP tools as first-class product capabilities" -Clienții diferiți ar trebui să aibă cel mai mic privilegiu de acces la categoriile de instrumente. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Cum o rezolvă OmniRoute:** +**How OmniRoute solves it:** -- 10 lunete MCP granulare pentru acces controlat la instrumente -- Aplicarea domeniului de aplicare și vizibilitatea în interfața de utilizare a managementului MCP -- Poziție implicită sigură pentru instrumentele operaționale
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding - -⚙️ 22. „Am nevoie de controale operaționale fără redistribuire” + -Echipele au nevoie de modificări rapide ale timpului de rulare în timpul incidentelor sau evenimentelor de cost. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**Cum o rezolvă OmniRoute:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Comutați activarea comboi direct din tabloul de bord MCP -- Aplicați profiluri de rezistență din pachetele de politici predefinite -- Resetați starea întreruptorului de la același panou de operare
+**How OmniRoute solves it:** - -🔄 23. „Am nevoie de vizibilitate și anulare a ciclului de viață a sarcinii A2A live” +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Fără vizibilitatea ciclului de viață, incidentele sarcinilor devin greu de triat. + -**Cum o rezolvă OmniRoute:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Listarea sarcinilor/filtrarea după stare/abilitate cu paginare -- Detaliați metadatele sarcinii, evenimentele și artefactele -- Punct final de anulare a sarcinii și acțiune UI cu confirmare
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. - -🌊 24. „Am nevoie de valori de flux active pentru încărcarea A2A” +**How OmniRoute solves it:** -Fluxurile de lucru în flux necesită o perspectivă operațională privind concurența și conexiunile live. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**Cum o rezolvă OmniRoute:** + -- Contoare active de flux integrate în starea A2A -- Marcaj de timp pentru ultima sarcină și numărătoare pentru fiecare stat -- Carduri de bord A2A pentru monitorizarea operațiunilor în timp real +
+📋 20. "I need auditable MCP tool execution" - -🪪 25. „Am nevoie de descoperire de agenți standard pentru clienți” +When tools mutate config or trigger ops actions, teams need forensic traceability. -Clienții externi și orchestratorii au nevoie de metadate care pot fi citite de mașină pentru integrare. +**How OmniRoute solves it:** -**Cum o rezolvă OmniRoute:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- Card de agent expus la `/.well-known/agent.json` -- Capabilități și abilități afișate în UI de management -- API-ul de stare A2A include metadate de descoperire pentru automatizare
+ - -🧭 26. „Am nevoie de descoperirea protocolului în UX-ul produsului” +
+🔐 21. "I need scoped MCP permissions per integration" -Dacă utilizatorii nu pot descoperi suprafețele de protocol, calitatea adoptării și a suportului scade. +Different clients should have least-privilege access to tool categories. -**Cum o rezolvă OmniRoute:** +**How OmniRoute solves it:** -- Pagina consolidată**Puncte finale**cu file pentru punctele finale Proxy, MCP, A2A și API -- Comută starea serviciului în linie (Online/Offline) pentru MCP și A2A -- Link-uri de la prezentare generală la file dedicate de gestionare
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling - -🧪 27. „Am nevoie de validarea protocolului end-to-end cu clienți reali” + -Testele simulate nu sunt suficiente pentru a valida compatibilitatea protocolului înainte de lansare. +
+⚙️ 22. "I need operational controls without redeploying" -**Cum o rezolvă OmniRoute:** +Teams need quick runtime changes during incidents or cost events. -- Suita E2E care pornește aplicația și utilizează transportul clientului MCP SDK real -- Testele client A2A pentru descoperirea, trimiterea, transmiterea în flux, obținerea și anularea fluxurilor -- Verificați încrucișați afirmațiile cu auditul MCP și API-urile pentru sarcini A2A
+**How OmniRoute solves it:** - -📡 28. „Am nevoie de observabilitate unificată pe toate interfețele” +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -Împărțirea observabilității în funcție de protocol creează puncte oarbe și MTTR mai lung. + -**Cum o rezolvă OmniRoute:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Tablouri de bord/jurnale/analitice unificate într-un singur produs -- Sănătate + audit + solicitare de telemetrie în straturi OpenAI, MCP și A2A -- API-uri operaționale pentru stare și automatizare
+Without lifecycle visibility, task incidents become hard to triage. - -💼 29. „Am nevoie de un timp de execuție pentru proxy + instrumente + orchestrare agent” +**How OmniRoute solves it:** -Rularea multor servicii separate crește costurile operaționale și modurile de eșec. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**Cum o rezolvă OmniRoute:** + -- Proxy compatibil OpenAI, server MCP și server A2A într-o singură stivă -- Autentificare partajată, rezistență, depozit de date și observabilitate -- Model de politică consistent pe toate suprafețele de interacțiune +
+🌊 24. "I need active stream metrics for A2A load" - -🚀 30. „Trebuie să trimit fluxuri de lucru agentice fără extinderea codului lipici” +Streaming workflows require operational insight into concurrency and live connections. -Echipele își pierd din viteza atunci când realizează mai multe servicii și scripturi ad-hoc. +**How OmniRoute solves it:** -**Cum o rezolvă OmniRoute:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Strategie unificată pentru clienți și agenți -- Interfețe de utilizare a protocolului încorporate și căi de validare a fumului -- Baze pregătite pentru producție (securitate, logare, rezistență, backup)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Playbook A: Maximizați abonamentul plătit + backup ieftin**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -607,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Playbook B: teanc de codare cu costuri zero**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Playbook C: lanț alternativ permanent activ 24/7**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -630,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Playbook D: Agentul operează cu MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Configurați codarea AI în minute la**$0/lună**. Conectați aceste conturi gratuite și utilizați combinația încorporată**Free Stack**. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Pasul | Acțiune | Furnizori deblocați | -| ---- | -------------------------------------------------- | ----------------------------------------------------------------- | -| 1 | Conectați**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**nelimitat**| -| 2 | Conectați**Qoder**(Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... —**nelimitat**| -| 3 | Conectați**Qwen**(Codul dispozitivului) | qwen3-coder-plus, qwen3-coder-flash... —**nelimitat**| -| 4 | Conectați**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180K/lună gratuit**| -| 5 | `/dashboard/combos` →**Stivă gratuită ($0)**șablon | Round-robin toți furnizorii gratuiti în mod automat | +| Step | Action | Providers Unlocked | +| ---- | -------------------------------------------------- | ------------------------------------------------------------------ | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Indicați orice IDE/CLI către:**`http://localhost:20128/v1` · Cheie API: `orice șir` · Terminat. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Acoperire suplimentară opțională (de asemenea gratuită):**Cheia API Groq (30 RPM gratuit), NVIDIA NIM (40 RPM fără, peste 70 de modele), Cerebras (1M tok/zi), LongCat API key (50M tokens/zi!), Cloudflare Workers AI (10K Neuroni/zi, 50+ modele).## Pornire rapidă +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Pornire rapidă ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **Utilizatori pnpm:**Rulați `pnpm approve-builds -g` după instalare pentru a activa scripturile de compilare native cerute de `better-sqlite3` și `@swc/core`: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > > ```bash > pnpm install -g omniroute -> pnpm approve-builds -g # Selectați toate pachetele → approve +> pnpm approve-builds -g # Select all packages → approve > omniroute > ``` -Tabloul de bord se deschide la `http://localhost:20128` și adresa URL de bază a API-ului este `http://localhost:20128/v1`. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Comanda | Descriere | -| ----------------------- | ----------------------------------------------------------------------- | -| `omniroute` | Porniți serverul (`PORT=20128`, API și tabloul de bord pe același port) | -| `omniroute --port 3000` | Setați portul canonic/API la 3000 | -| `omniroute --mcp` | Porniți serverul MCP (transport stdio) | -| `omniroute --no-open` | Nu deschideți automat browserul | -| `omniroute --help` | Arată ajutor | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Opțional modul split-port:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -Pentru majoritatea implementărilor, aveți nevoie doar de: +For most deployments, you only need: -| Variabila | Implicit | Scop | -| ------------------------ | ----------------------------- | ------------------------------------------------------------ ------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600000` | Linie de referință partajată pentru preluarea în amonte, timeout-uri ascunse Undici, solicitări de amprentă TLS și timeout-uri de solicitare/proxy bridge API | -| `STREAM_IDLE_TIMEOUT_MS` | moștenește `REQUEST_TIMEOUT_MS` | Distanța maximă între bucățile de streaming înainte ca OmniRoute să anuleze fluxul SSE | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -Compatibilitatea cu versiunea anterioară este păstrată: `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` existente și alte variante de expirare pe strat încă funcționează și suprascriu linia de bază partajată. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Sunt disponibile modificări avansate dacă aveți nevoie de un control mai fin:| Variabila | Implicit | Scop | -| ---------------------------------------- | ------------------------------------------ | ------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | moștenește `REQUEST_TIMEOUT_MS` | Timeout total de solicitare în amonte utilizat de semnalul principal de întrerupere a preluării | -| `FETCH_HEADERS_TIMEOUT_MS` | moștenește `FETCH_TIMEOUT_MS` | Limită de timp Undici pentru primirea antetelor de răspuns în amonte | -| `FETCH_BODY_TIMEOUT_MS` | moștenește `FETCH_TIMEOUT_MS` | Limita de timp Undici între bucățile de corp din amonte (`0` îl dezactivează) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP conectare timeout | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | -| `TLS_CLIENT_TIMEOUT_MS` | moștenește `FETCH_TIMEOUT_MS` | Timeout pentru solicitările de amprentă TLS făcute prin `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | moștenește `REQUEST_TIMEOUT_MS` sau `30000` | Timeout pentru redirecționarea proxy `/v1` de la portul API la portul tabloului de bord | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Solicitarea de intrare expiră pe serverul API bridge | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Timeout antet de intrare pe serverul API bridge | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Timeout de menținere activă pe serverul API bridge | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Timeout inactivitate socket pe serverul de punte API (`0` îl dezactivează) | +Advanced overrides are available if you need finer control: -Dacă rulați OmniRoute în spatele Nginx, Caddy, Cloudflare sau alt proxy invers, asigurați-vă că proxy-ul -timeout-urile sunt, de asemenea, mai mari decât timeout-urile pentru flux/preluare OmniRoute.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Deschideți Dashboard → `Providers` și conectați cel puțin un furnizor (OAuth sau cheie API). -2. Deschideți Dashboard → `Endpoints` și creați o cheie API. -3. (Opțional) Deschideți Dashboard → `Combos` și setați lanțul de rezervă.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Funcționează cu Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode și SDK-uri compatibile cu OpenAI.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (pentru operațiuni cu scule):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -Apoi conectați-vă clientul MCP la `stdio` și instrumente de testare precum: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (pentru fluxuri de lucru de la agent la agent):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -759,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Această suită validează fluxurile reale de clienți MCP și A2A împotriva unei aplicații care rulează.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -767,13 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` - -Void Linux (șablon `xbps-src`) +
+Void Linux (`xbps-src` template) -Pentru utilizatorii Void Linux, puteți construi un pachet nativ folosind `xbps-src`. Salvați acest bloc ca `srcpkgs/omniroute/template`:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -785,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -793,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -869,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -880,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute este disponibil ca imagine publică Docker pe [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Alergare rapidă:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -890,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**Cu fișierul de mediu:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Utilizarea Docker Compose:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Suportul tabloului de bord pentru implementările Docker include acum un singur clic**Cloudflare Quick Tunnel**pe `Tabloul de bord → Puncte finale`. Prima activare descarcă `cloudflare` numai când este necesar, pornește un tunel temporar către punctul final `/v1` curent și afișează adresa URL generată `https://*.trycloudflare.com/v1` direct sub adresa URL publică normală. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Note: +Notes: -- URL-urile tunelului rapid sunt temporare și se modifică după fiecare repornire. -- Tunelurile rapide nu sunt restaurate automat după repornirea unui OmniRoute sau a containerului. Reactivați-le din tabloul de bord când este necesar. -- Instalarea gestionată acceptă în prezent Linux, macOS și Windows pe `x64` / `arm64`. -- Managed Quick Tunnels este implicit la transportul HTTP/2 pentru a evita avertismentele zgomotoase ale bufferului QUIC UDP în mediile de containere constrânse. Setați `CLOUDFLARED_PROTOCOL=quic` sau `auto` dacă doriți un alt transport. -- Imaginile Docker reunesc rădăcinile CA ale sistemului și le transmit la „cloudflared” gestionat, ceea ce evită eșecurile de încredere TLS atunci când tunelul pornește în interiorul containerului. -- SQLite rulează în modul WAL. `Docker stop` ar trebui permis să se termine, astfel încât OmniRoute să poată verifica cele mai recente modificări înapoi în `storage.sqlite`. -- Fișierele Compose incluse au stabilit deja o perioadă de grație de 40 de ani. Dacă rulați imaginea direct, păstrați `--stop-timeout 40` (sau similar), astfel încât opririle manuale să nu întrerupă curățarea închiderii. -- Setați `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` dacă doriți ca OmniRoute să folosească un binar existent în loc să descarce unul. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Utilizarea Docker Compose cu Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute poate fi expus în siguranță utilizând furnizarea automată SSL de la Caddy. Asigurați-vă că înregistrarea DNS A a domeniului dvs. indică IP-ul serverului dvs.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` - -| Imagine | Etichetă | Dimensiune | Descriere | +| Image | Tag | Size | Description | | ------------------------ | -------- | ------ | --------------------- | -| `diegosouzapw/omniroute` | `ultimul` | ~250MB | Ultima versiune stabilă | -| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Versiunea curentă |--- +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | + +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**NOU!**OmniRoute este acum disponibil ca**aplicație desktop nativă**pentru Windows, macOS și Linux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Rulați OmniRoute ca o aplicație desktop autonomă - fără terminal, fără browser, fără internet necesar pentru modelele locale. Aplicația bazată pe electroni include: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Fereastra nativă**— Fereastra aplicației dedicată cu integrare în tava de sistem -- 🔄**Auto-Start**— Lansați OmniRoute la autentificarea sistemului -- 🔔**Notificări native**— Primiți alerte pentru epuizarea cotelor sau probleme legate de furnizor -- ⚡**Instalare cu un singur clic**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Mod offline**— Funcționează complet offline cu serverul inclus### Pornire rapidă +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Pornire rapidă ```bash # Development mode @@ -979,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -Când este minimizat, OmniRoute se află în bara de sistem cu acțiuni rapide: +When minimized, OmniRoute lives in your system tray with quick actions: -- Deschide tabloul de bord -- Schimbați portul serverului -- Închideți aplicația +- Open dashboard +- Change server port +- Quit application -📖 Documentație completă: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Nivelul | Furnizor | Cost | Resetare cotă | Cel mai bun pentru | -| ---------------- | --------------------------- | ------------------------------------ | --------------------------- | --------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -| **💳 ABONARE** | Claude Code (Pro) | 20 USD/lună | 5h + săptămânal | Deja abonat | -| | Codex (Plus/Pro) | 20-200 USD/lună | 5h + săptămânal | Utilizatori OpenAI | -| | Gemeni CLI | **GRATIS** | 180K/lună + 1K/zi | Toată lumea! | -| | GitHub Copilot | 10-19 USD/lună | Lunar | utilizatorii GitHub | -| **🔑 CHEIA API** | NVIDIA NIM | **GRATIS**(dev forever) | ~40 RPM | 70+ modele deschise | -| | Cerebre | **GRATIS**(1M tok/zi) | 60K TPM / 30 RPM | Cel mai rapid din lume | -| | Groq | **GRATIS**(30 RPM) | 14,4K RPD | Llama/Gemma ultra-rapidă | -| | DeepSeek V3.2 | 0,27 USD/1,10 USD per 1 milion | Niciuna | Cel mai bun raționament preț/calitate | -| | xAI Grok-4 Fast | **0,20 USD/0,50 USD pe 1M**🆕 | Niciuna | Cea mai rapidă + apelare instrument, ultralow | -| | xAI Grok-4 (standard) | 0,20 USD/1,50 USD per 1 milion 🆕 | Niciuna | Raționamentul emblematic de la xAI | -| | Mistral | Probă gratuită + plătit | Tarif limitat | IA europeană | -| | OpenRouter | Plată-pe-utilizare | Niciuna | 100+ modele agr. | -| **💰 IEFTIN** | GLM-5 (prin Z.AI) 🆕 | 0,5 USD/1 milion | Zilnic 10:00 | Ieșire 128K, cel mai nou flagship | -| | GLM-4.7 | 0,6 USD/1 milion | Zilnic 10:00 | Backup buget | -| | MiniMax M2.5 🆕 | Intrare de 0,3 USD/1 milion | rulare de 5 ore | Raționament + sarcini agentice | -| | MiniMax M2.1 | 0,2 USD/1 milion | rulare de 5 ore | Cea mai ieftină opțiune | -| | Kimi K2.5 (API Moonshot) 🆕 | Plată-pe-utilizare | Niciuna | Acces direct API Moonshot | -| | Kimi K2 | 9 USD/lună plat | 10 milioane de jetoane/lună | Cost previzibil | -| **🆓 GRATUIT** | Qoder | **$0** | Nelimitat | 5 modele nelimitat | -| | Qwen | **$0** | Nelimitat | 4 modele nelimitat | -| | Kiro | **$0** | Nelimitat | Claude Sonnet/Haiku (AWS Builder) | -| | LongCat Flash-Lite 🆕 | **$0**(50 M tok/zi 🔥) | 1 RPS | Cea mai mare cotă gratuită de pe Pământ | -| | Polenizări AI 🆕 | **$0**(nu este nevoie de cheie) | 1 solicitat/15s | GPT-5, Claude, DeepSeek, Llama 4 | -| | Cloudflare Workers AI 🆕 | **$0**(10K Neuroni/zi) | ~150 resp/zi | 50+ modele, avantaj global | -| | Scaleway AI 🆕 | **$0**(1 milion de jetoane în total) | Tarif limitat | UE/GDPR, Qwen3 235B, Llama 70B | > 🆕**Modele noi adăugate (mar 2026):**Familia Grok-4 Fast la 0,20 USD/0,50 USD/M (evaluat la 1143 ms — 30% mai rapid decât Gemini 2.5 Flash), GLM-5 prin Z.AI cu ieșire de 128K, MiniMax M2.5 Raționamentul, actualizarea KimiSeek V3. API direct Moonshot. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 Stivă combinată de 0 USD — Configurare completă gratuită:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Cost zero. Nu se oprește niciodată codificarea.**Configurați acest lucru ca un combo OmniRoute și toate alternativele au loc automat - fără comutare manuală vreodată.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Toate modelele de mai jos sunt**100% gratuite, fără card de credit necesar**. OmniRoute face trasee automate între ele atunci când se epuizează o cotă - combină-le pe toate pentru o combinație de 0 USD de neîntrerupt.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Model | Prefix | Limită | Limită de rată | +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | --------------------- | -| `claude-sonet-4.5` | `kr/` |**Nelimitat**| Niciun plafon zilnic raportat | -| `claude-haiku-4.5` | `kr/` |**Nelimitat**| Niciun plafon zilnic raportat | -| `claude-opus-4.6` | `kr/` |**Nelimitat**| Ultimul Opus prin Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -| Model | Prefix | Limită | Limită de rată | +### 🟢 QODER MODELS (Free PAT via qodercli) + +| Model | Prefix | Limit | Rate Limit | | ------------------ | ------ | ------------- | --------------- | -| `kimi-k2-thinking` | `dacă/` |**Nelimitat**| Nicio limită raportată | -| `qwen3-coder-plus` | `dacă/` |**Nelimitat**| Nicio limită raportată | -| `deepseek-r1` | `dacă/` |**Nelimitat**| Nicio limită raportată | -| `minimax-m2.1` | `dacă/` |**Nelimitat**| Nicio limită raportată | -| `kimi-k2` | `dacă/` |**Nelimitat**| Nicio limită raportată | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -> Metoda de conectare recomandată:**Personal Access Token + `qodercli`**. Browser OAuth este -> experimental și dezactivat implicit, cu excepția cazului în care variabilele de mediu `QODER_OAUTH_*` sunt configurate.### 🟡 QWEN MODELS (Device Code Auth) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| Model | Prefix | Limită | Limită de rată | +### 🟡 QWEN MODELS (Device Code Auth) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | ------------------- | -| `qwen3-coder-plus` | `qw/` |**Nelimitat**| Nicio limită raportată | -| `qwen3-coder-flash` | `qw/` |**Nelimitat**| Nicio limită raportată | -| `qwen3-coder-next` | `qw/` |**Nelimitat**| Nicio limită raportată | -| `model-vizual` | `qw/` |**Nelimitat**| Multimodal (imagini) |### 🟣 GEMINI CLI (Google OAuth) +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Model | Prefix | Limită | Limită de rată | +### 🟣 GEMINI CLI (Google OAuth) + +| Model | Prefix | Limit | Rate Limit | | ------------------------ | ------ | --------------------------- | ------------- | -| `gemeni-3-flash-preview` | `gc/` |**180K tok/lună**+ 1K/zi | Resetare lunară | -| `gemini-2.5-pro` | `gc/` | 180K/lună (piscina comună) | Calitate înaltă |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -| Nivelul | Limită zilnică | Limită de rată | Note | +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---------- | ------------ | ----------- | ------------------------------------------------------ | -| Gratuit (Dev) | Fără capac de simbol |**~40 RPM**| 70+ modele; trecerea la limitele ratei pure la mijlocul anului 2025 | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -Modele gratuite populare: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-deepseek`,/deepseek`,/deepseek-70b### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -| Nivelul | Limită zilnică | Limită de rată | Note | +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) + +| Tier | Daily Limit | Rate Limit | Notes | | ---- | ----------------- | ---------------- | ------------------------------------------- | -| Gratuit |**1 milion de jetoane/zi**| 60K TPM / 30 RPM | Cea mai rapidă inferență LLM din lume; resetează zilnic | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -Disponibil gratuit: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -| Nivelul | Limită zilnică | Limită de rată | Note | +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---- | ------------- | ---------------- | ----------------------------------------- | -| Gratuit |**14,4K RPD**| 30 RPM per model | Fără card de credit; 429 la limită, netaxat | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -Disponibil gratuit: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -| Model | Prefix | Cotă zilnică gratuită | Note | +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 + +| Model | Prefix | Daily Free Quota | Notes | | ----------------------------- | ------ | ----------------- | ----------------------- | -| `LongCat-Flash-Lite` | `lc/` |**50M de jetoane**💥 | Cea mai mare cotă gratuită vreodată | -| `LongCat-Flash-Chat` | `lc/` | 500K jetoane | Chat cu mai multe ture | -| `LongCat-Flash-Thinking` | `lc/` | 500K jetoane | Raționament / CoT | -| `LongCat-Flash-Thinking-2601` | `lc/` | 500K jetoane | Versiunea ianuarie 2026 | -| `LongCat-Flash-Omni-2603` | `lc/` | 500K jetoane | Multimodal | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | -> 100% gratuit în timpul beta public. Înscrieți-vă la [longcat.chat](https://longcat.chat) cu e-mail sau telefon. Se resetează zilnic la 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. -| Model | Prefix | Limită de rată | Furnizor în spatele | +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| `openai` | `pol/` | 1 solicitat/15s | GPT-5 | -| `claude` | `pol/` | 1 solicitat/15s | Claude antropic | -| `gemeni` | `pol/` | 1 solicitat/15s | Google Gemeni | -| `deepseek` | `pol/` | 1 solicitat/15s | DeepSeek V3 | -| `llama` | `pol/` | 1 solicitat/15s | Meta Llama 4 Scout | -| `mistral` | `pol/` | 1 solicitat/15s | Mistral AI | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**Zero frecare:**Fără înscriere, fără cheie API. Adăugați furnizorul de polenizări cu un câmp cheie gol și funcționează imediat.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| Nivelul | Neuroni zilnici | Utilizare echivalentă | Note | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 + +| Tier | Daily Neurons | Equivalent Usage | Notes | | ---- | ------------- | --------------------------------------- | ----------------------- | -| Gratuit |**10.000**| ~150 LLM resp / 500s audio / 15K încorporare | Avantaj global, peste 50 de modele | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -Modele gratuite populare: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (audio gratuit!), `@cf/qwen/qwen2.5-coder-15b-instruct` +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -> Necesită API Token + ID cont de la [dash.cloudflare.com](https://dash.cloudflare.com). Stocați ID-ul contului în setările furnizorului.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -| Nivelul | Cotă gratuită | Localizare | Note | +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | | ---- | ------------- | ------------ | ----------------------------------- | -| Gratuit |**1 milion de jetoane**| 🇫🇷 Paris, UE | Nu este nevoie de card de credit în limite | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | -Disponibil gratuit: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` -> Conform UE/GDPR. Obțineți cheia API la [console.scaleway.com](https://console.scaleway.com). +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). ->**💡 Ultima stivă gratuită (11 furnizori, 0 USD pentru totdeauna):** +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Sonnet/Haiku NELIMITAT -> Qoder (dacă/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 NELIMITAT -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 de milioane de jetoane/zi 🔥 -> Polenizări (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — nu este necesară cheia -> Qwen (qw/) → modele qwen3-coder NELIMITAT -> Gemeni (gemeni/) → Gemeni 2.5 Flash — 1.500 de solicitări/zi gratuit -> Cloudflare AI (cf/) → 50+ modele — 10K neuroni/zi -> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1 milion de jetoane gratuite (UE) -> Groq (groq/) → Llama/Gemma — 14.4K solicitat/zi ultra-rapid -> NVIDIA NIM (nvidia/) → 70+ modele deschise — 40 RPM pentru totdeauna -> Cerebre (cerebre/) → Llama/Qwen cel mai rapid din lume — 1M tok/zi -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Transcrie orice audio/video pentru**$0**— Deepgram conduce cu 200 USD gratuit, AssemblyAI 50 USD alternativ, Groq Whisper ca rezervă nelimitată de urgență. +## 🎙️ Free Transcription Combo -| Furnizor | Credite gratuite | Cel mai bun model | Limită de rată | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. + +| Provider | Free Credits | Best Model | Rate Limit | | ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | -| 🟢**Deepgram**|**200 USD gratuit**(înscriere) | `nova-3` — cea mai bună acuratețe, peste 30 de limbi | Fără limită RPM pentru creditele gratuite | -| 🔵**AsamblareAI**|**50 USD gratuit**(înscriere) | `universal-3-pro` — capitole, sentiment, PII | Fără limită RPM pentru creditele gratuite | -| 🔴**Groq**|**Gratuit pentru totdeauna**| `whisper-large-v3` — OpenAI Whisper | 30 RPM (rată limitată) | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | -**Combinație sugerată în `/dashboard/combos`:**``` +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Apoi, în `/dashboard/media` → fila**Transcriere**: încărcați orice fișier audio sau video → selectați punctul final combo → obțineți transcrierea în formatele acceptate.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 este construit ca o platformă operațională, nu doar un proxy releu.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Caracteristica | Ce face | -| -------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Grok-4 Fast Family** | Modele xAI la 0,20 USD/0,50 USD/M — etalon de 1143 ms (30% mai rapid decât Gemini 2.5 Flash) | -| 🧠**GLM-5 prin Z.AI** | Context de ieșire de 128.000, 0,5 USD/1 milion — cel mai nou flagship din familia GLM | -| 🔮**MiniMax M2.5** | Raționament + sarcini agentice la 0,30 USD/1 milion – upgrade semnificativ de la M2.1 | -| 🎯**toolCalling Flag per model** | „ToolCalling: true/false” per model în registru — AutoCombo omite modelele care nu sunt compatibile cu instrumente | -| 🌍**Detecția intenției multilingve** | Cuvinte cheie PT/ZH/ES/AR în scorul AutoCombo — o selecție mai bună a modelului pentru conținut care nu este în limba engleză | -| 📊**Backmark-uri bazate pe benchmark** | Latența reală p95 de la solicitările live alimentează scorul combinat — AutoCombo învață din datele reale | -| 🔁**Solicitare deduplicare** | Fereastra de deduplicare bazată pe hash de conținut — sigură pentru mai mulți agenți, previne taxele duplicate | -| 🔌**Pluggable RouterStrategy** | Interfață extensibilă `RouterStrategy` — adăugați logica de rutare personalizată ca pluginuri | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Caracteristica | Ce face | -| ------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Teren de joacă model** | Pagina tabloului de bord pentru a testa orice model direct — selectoare furnizor/model/punct final, Editor Monaco, streaming, anulare, sincronizare | -| 🔏**Potrivirea amprentei CLI** | Ordinea antetului/corpului pentru fiecare furnizor pentru a se potrivi cu semnăturile CLI native - comutați pentru fiecare furnizor în Setări > Securitate.**IP-ul dvs. proxy este păstrat** | -| 🤝**Suport ACP (Agent Client Protocol)** | Descoperire agent CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw + încă 9), generator de proces, punct final `/api/acp/agents` | -| 🤖**Tabloul de bord pentru agenții ACP** | Depanare › Pagina Agenți — grilă de 14 agenți cu starea instalării, versiunea, formularul de agent personalizat pentru orice instrument CLI. Utilizatorii**OpenCode**primesc un buton „Descărcați opencode.json” care generează automat o configurație gata de utilizare cu toate modelele disponibile. | -| 🔧**Rutare `apiFormat` model personalizat** | Modelele personalizate cu `apiFormat: "răspunsuri"` sunt acum direcționate corect către traducătorul API-ului Responses | -| 🏢**Izolarea spațiului de lucru Codex** | Spații de lucru Codex multiple per e-mail — OAuth separă corect conexiunile după ID-ul spațiului de lucru | -| 🔄**Actualizare automată Electron** | Aplicația desktop verifică actualizările + instalare automată la repornire | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Caracteristica | Ce face | -| -------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**Server MCP (25 de instrumente)** | Instrumente IDE/agent prin 3 transporturi: stdio, SSE (`/api/mcp/sse`), HTTP Streamable (`/api/mcp/stream`). 18 nuclee + 3 memorie + 4 instrumente de calificare | -| 🤝**Server A2A (JSON-RPC + SSE)** | Execuția sarcinilor de la agent la agent cu fluxuri de sincronizare și streaming | -| 🧭**Pagină de puncte finale consolidate** | Pagina de gestionare cu file cu file Endpoint Proxy, MCP, A2A și API Endpoints | -| 🎚️**Servicii Activare/Dezactivare Comutări** | Comutatoare ON/OFF pentru MCP și A2A cu persistența setărilor (implicit: OFF) | -| 🛰️**MCP Runtime Heartbeat** | Starea reală a procesului (pid, uptime, vârsta bătăilor inimii, transport, mod scope) | -| 📋**MCP Audit Trail** | Jurnale de audit filtrabile cu succes/eșec și atribuire cheie | -| 🔐**MCP Scope Enforcement** | 10 permisiuni granulare pentru acces controlat la instrument | -| 📡**A2A Task Lifecycle Management** | Listați/filtrați sarcinile, inspectați evenimentele/artefactele, anulați activitățile care rulează | -| 📋**Descoperire card de agent** | `/.well-known/agent.json` pentru descoperirea automată a clientului | -| 🧪**Protocol E2E Test Harness** | Real MCP SDK + A2A client flux în `test:protocoale:e2e` | -| ⚙️**Controale operaționale** | Combo de comutare, aplicați profile de rezistență, resetați întrerupătoarele de pe o suprafață de control | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Caracteristica | Ce face | -| ---------------------------------------------- | ----------------------------------------------------------------------------------------- | ----------------------- | -| 🎯**Backback inteligent pe 4 niveluri** | Rută automată: Abonament → Cheie API → Ieftin → Gratuit | -| 📊**Urmărirea cotelor în timp real** | Numărătoare de jetoane live + numărătoare inversă de resetare per furnizor | -| 🔄**Traducerea formatului** | OpenAI ↔ Claude ↔ Gemeni ↔ Răspunsuri cu conversii sigure pentru schema | -| 👥**Asistență pentru mai multe conturi** | Conturi multiple per furnizor cu selecție inteligentă | -| 🔄**Reîmprospătare automată a simbolului** | Tokenurile OAuth se reîmprospătează automat cu reîncercarea | -| 🎨**Combinații personalizate** | 9 strategii de echilibrare + controlul lanțului de rezervă | -| 🌐**Wildcard Router** | `furnizor/*` rutare dinamică | -| 🧠**Gândirea controalelor bugetare** | Limite de raționament de trecere, automate, personalizate și adaptive | -| 🔀**Aliasuri de model** | Aliasarea modelului încorporat + personalizat și siguranța migrării | -| ⚡**Degradarea fundalului** | Direcționați sarcinile de fundal cu prioritate redusă către modele mai ieftine | -| 🧪**Rutare inteligentă în funcție de sarcini** | Se selectează automat modelul după tipul de conținut (codificare/viziune/analiza/rezumat) | -| 🔄**Fluxuri de lucru pentru agenți A2A** | Orchestrator determinist FSM pentru execuții de agenți cu mai mulți pași | -| 🔀**Rutare adaptivă** | Modificarea dinamică a strategiei bazată pe volumul jetonului și complexitatea promptă | -| 🎲**Diversitatea furnizorilor** | Scorul entropiei Shannon echilibrând distribuția traficului combo automat | -| 💬**System Prompt Injection** | Controalele globale ale comportamentului aplicate în mod consecvent | -| 📄**Responses API Compatibility** | Suport complet `/v1/responses` pentru Codex și fluxuri de lucru agentice avansate | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Caracteristica | Ce face | -| ------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ---------------------------------------- | -| 🖼️**Generarea imaginii** | `/v1/images/generations` cu cloud și backend-uri locale | -| 📐**Inglobări** | `/v1/embeddings` pentru conducte de căutare și RAG | -| 🎤**Transcriere audio** | `/v1/audio/transcriptions` — 7 furnizori (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), detectarea automată a limbii, suport MP4/MP3/WAV | -| 🔊**Text-to-speech** | `/v1/audio/speech` — 10 furnizori (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) cu mesaje de eroare corecte | -| 🎬**Generație video** | `/v1/videos/generations` (fluxuri de lucru ComfyUI + SD WebUI) | -| 🎵**Generație muzicală** | `/v1/music/generations` (fluxuri de lucru ComfyUI) | -| 🛡️**Moderații** | verificări de siguranță `/v1/moderations` | -| 🔀**Reclasificare** | `/v1/rerank` pentru scorul de relevanță | -| 🔍**Căutare pe web**🆕 | `/v1/search` — 5 furnizori (Serper, Brave, Perplexity, Exa, Tavily), peste 6.500 gratuit/lună, auto-failover, cache | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Caracteristica | Ce face | -| --------------------------------------------- | --------------------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**Întrerupătoare** | Deplasare/recuperare per model cu controale de prag | -| 🎯**Modele compatibile cu punctele finale** | Modelele personalizate declară puncte finale acceptate + format API | -| 🛡️**Turmă Anti-Tunete** | Protecții Mutex + semafor la evenimentele de reîncercare/evaluare | -| 🧠**Semantică + Cache de semnătură** | Reducerea costului/latenței cu două straturi de cache | -| ⚡**Solicita Idempotenta** | Fereastra de protecție duplicată | -| 🔒**TLS Fingerprint Spoofing** | Amprenta digitală TLS asemănătoare unui browser —**reduce detectarea botului și semnalizarea contului** | -| 🔏**Potrivirea amprentei CLI** | Se potrivește cu semnăturile cererilor CLI native —**reduce riscul de interzicere, păstrând IP-ul proxy** | -| 🌐**Filtrare IP** | Controlul listei de permise/liste de blocare pentru implementările expuse | -| 📊**Limite de rată editabile** | Limite configurabile globale/la nivel de furnizor cu persistență | -| 📉**Degradare grațioasă** | Funcții de rezervă a capacității multistrat care protejează operațiunile de bază ale gateway-ului | -| 📜**Config Audit Trail** | Urmărirea modificărilor bazată pe diferențe prevenind deviația operațională prin derulări simple | -| ⏳**Sincronizarea sănătății furnizorului** | Monitorizarea proactivă a expirării jetonului care declanșează alerte înainte de eșecurile de autorizare | -| 🚪**Dezactivați automat conturile interzise** | Întrerupător de circuit operațional sigilând automat conturile de simbol blocate permanent | -| 🔑**Administrarea cheilor API + Scoping** | Emiterea/rotarea cheilor securizate și controale model/furnizor | -| 👁️**Scoped API Key Reveal**🆕 | Înscrieți-vă recuperarea cheilor API prin `ALLOW_API_KEY_REVEAL` | -| 🛡️**Protejat `/modele`** | Autentificare opțională și ascunderea furnizorului pentru catalogul de modele | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Caracteristica | Ce face | -| ------------------------------------------------ | ---------------------------------------------------------------------------------- | ---------------------------- | -| 📝**Solicitare + Înregistrare proxy** | Cerere/răspuns complet și înregistrare proxy | -| 📉**Jurnalele detaliate în flux**🆕 | Reconstituie fluxurile de sarcină utilă SSE în mod curat în UI | -| 📋**Tabloul de bord pentru jurnalele unificate** | Vizualizări de solicitare, proxy, audit și consolă într-o singură pagină | -| 🔍**Solicitare telemetrie** | latența p50/p95/p99 și urmărirea solicitărilor | -| 🏥**Tabloul de bord pentru sănătate** | Uptime, stări de întrerupere, blocări, statistici cache | -| 💰**Urmărirea costurilor** | Controalele bugetare și vizibilitatea prețurilor pe model | -| 📈**Vizualizări de analiză** | Informații despre utilizarea modelului/furnizorului și vizualizări ale tendințelor | -| 🧪**Cadru de evaluare** | Testarea setului de aur cu strategii de meci configurabile | -| 📡**Diagnosticare live**🆕 | Ocolire semantică a memoriei cache pentru testare combo în direct precisă | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Caracteristica | Ce face | -| ------------------------------------ | ------------------------------------------------------------------------------ | --------------------- | -| 🌐**Implementează oriunde** | Localhost, VPS, Docker, medii cloud | -| 🚇**Cloudflare Tunnel**🆕 | Integrare rapidă a tunelului cu un singur clic din tabloul de bord | -| 🔑**Filtrare model cheie API** | Răspunsul nativ /v1/models filtrat prin rolurile de context Bearer atribuite | -| ⚡**Smart Cache Bypass** | Euristice TTL configurabile și comenzi de recăpătare forțată | -| 🔄**Backup/Restaurare** | Export/import și fluxuri de recuperare în caz de dezastru | -| 🧙**Onboarding Wizard** | Configurare ghidată pentru prima rulare | -| 🔧**Tabloul de bord CLI Tools** | Configurare cu un singur clic pentru instrumentele populare de codare | -| 🎮**Teren de joacă model** | Testați orice furnizor/model/punct final din tabloul de bord | -| 🔏**CLI Fingerprint Toggle** | Potrivirea amprentelor pentru fiecare furnizor în Setări > Securitate | -| 🌐**i18n (30 de limbi)** | Tabloul de bord complet + suport pentru limbajul documentelor cu acoperire RTL | -| 🧹**Șterge toate modelele** | Ștergerea listei de modele cu un singur clic în detaliile furnizorului | -| 👁️**Comenzi din bara laterală**🆕 | Ascunde componentele și integrările din Setări de aspect | -| 📋**Șabloane de probleme** | Șabloane GitHub standardizate pentru erori și caracteristici | -| 📂**Director de date personalizate** | `DATA_DIR` înlocuire pentru locația de stocare | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1292,103 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -Când cota, rata sau starea de sănătate eșuează, OmniRoute trece automat la următorul candidat fără comutare manuală.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A pot fi descoperite în UI și în documente (nu sunt ascunse) -- API-urile privind starea protocolului expun date operaționale live (`/api/mcp/*`, `/api/a2a/*`) -- Tablourile de bord includ acțiuni pentru operațiunile din a doua zi (comutații combo, resetări întrerupătoare, anularea sarcinilor)#### Translator + validation workflow +#### Protocol management that is visible and operable -Zona Translator include: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Teren de joacă**: solicitați verificări de transformare -**Tester de chat**: cerere/răspuns complet dus-întors -**Bancul de testare**: mai multe cazuri într-o singură cursă -**Live Monitor**: vizualizare în timp real a traficului +#### Translator + validation workflow -Plus validarea protocolului cu clienți reali prin `npm run test:protocols:e2e`. +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Referință pentru instrumente, configurații IDE și exemple de clienți +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A Server README](src/lib/a2a/README.md)**— Abilități, metode JSON-RPC, streaming și ciclul de viață al sarcinilor## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute include un cadru de evaluare încorporat pentru a testa calitatea răspunsului LLM față de un set de aur. Accesați-l prin**Analitice → Evaluări**în tabloul de bord.### Built-in Golden Set +## 🧪 Evaluations (Evals) -„Setul de aur OmniRoute” preîncărcat conține cazuri de testare pentru: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Salutări, matematică, geografie, generare de cod -- Conformitatea formatului JSON, traducere, generare de reduceri -- Refuz de siguranță (conținut nociv), numărare, logică booleană### Evaluation Strategies +### Built-in Golden Set -| Strategie | Descriere | Exemplu | -| -------------- | ------------------------------------------------------------------------- | --------------------------------- | --- | -| `exact` | Ieșirea trebuie să se potrivească exact cu | `"4"` | -| `conține` | Ieșirea trebuie să conțină subșir (indiferență de majuscule și minuscule) | `"Paris"` | -| `regex` | Ieșirea trebuie să se potrivească cu modelul regex | `"1.*2.*3"` | -| `personalizat` | Funcția JS personalizată returnează adevărat/fals | `(ieșire) => ieșire.lungime > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) - -🧩 Configurare MCP (Model Context Protocol) +
+🧩 MCP Setup (Model Context Protocol) -Porniți transportul MCP în modul stdio:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Flux de validare recomandat: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Conectați clientul MCP prin stdio. -2. Rulați `omniroute_get_health`. -3. Rulați `omniroute_list_combos`. -4. Deschideți `/dashboard/mcp` pentru a confirma bătăile inimii, activitatea și auditul. - -API-uri utile pentru automatizare: +Useful APIs for automation: - `GET /api/mcp/status` - `GET /api/mcp/tools` - `GET /api/mcp/audit` -- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats` - -🤝 Configurare A2A (Agent2Agent) + -Descoperiți agentul:```bash +
+🤝 A2A Setup (Agent2Agent) + +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Trimiteți o sarcină:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` - -Gestionați ciclul de viață: +Manage lifecycle: - `GET /api/a2a/status` - `GET /api/a2a/tasks` - `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -Interfața de utilizare operațională: +Operational UI: -- `/dashboard/a2a` pentru observabilitatea sarcinii/starea/fluxului și acțiunilor de fum
+- `/dashboard/a2a` for task/state/stream observability and smoke actions - -🧪 Validare end-to-end a protocolului + -Validați ambele protocoale cu clienți reali:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Aceasta verifică: +This verifies: -- Conectare/lista/apelare client MCP SDK -- A2A descoperire/trimitere/stream/obține/anulează -- Verificați încrucișați datele în auditul MCP și API-urile de gestionare a sarcinilor A2A
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs - -💳 Furnizori de abonament### Claude Code (Pro/Max) + + +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1401,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Sfat profesionist:**Folosiți Opus pentru sarcini complexe, Sonnet pentru viteză. OmniRoute urmărește cota per model!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1415,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Fiecare cont Codex are acum comutări de politică în „Tabloul de bord -> Furnizori”: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (ON/OFF): aplicați politica privind pragul ferestrei de 5 ore. -- `Săptămânal` (ON/OFF): impuneți politica săptămânală privind pragul ferestrei. -- Comportamentul de prag: când o fereastră activată atinge >=90% utilizare, acel cont este omis. -- Comportament de rotație: OmniRoute direcționează automat către următorul cont Codex eligibil. -- Comportament de resetare: când furnizorul `resetAt` trece timpul, contul devine din nou eligibil automat. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Scenarii: +Scenarios: -- `5h ON` + `Weekly ON`: contul este omis atunci când oricare dintre ferestre atinge pragul. -- `5h OFF` + `Săptămânal ON`: numai utilizarea săptămânală poate bloca contul. -- `5h ON` + `Săptămânal OFF`: doar utilizarea de 5 ore poate bloca contul. -- `resetAt` a trecut: contul reintră automat în rotație (fără reactivare manuală).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1440,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Cea mai bună valoare:**Nivel gratuit imens! Utilizați acest lucru înainte de nivelurile plătite.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1455,71 +1662,91 @@ Models:
- -🔑 Furnizori cheie API### NVIDIA NIM (FREE developer access — 70+ models) +
+🔑 API Key Providers -1. Înscrieți-vă: [build.nvidia.com](https://build.nvidia.com) -2. Obțineți cheia API gratuită (1000 de credite de inferență incluse) -3. Tabloul de bord → Adăugați furnizor → NVIDIA NIM: - - Cheie API: `nvapi-your-key` +### NVIDIA NIM (FREE developer access — 70+ models) -**Modele:**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` și peste 50 de altele +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Sfat profesional:**API compatibil cu OpenAI - funcționează perfect cu traducerea formatului OmniRoute!### DeepSeek +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -1. Înscrieți-vă: [platform.deepseek.com](https://platform.deepseek.com) -2. Obțineți cheia API -3. Tabloul de bord → Adăugați furnizor → DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -**Modele:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +### DeepSeek -1. Înscrieți-vă: [console.groq.com](https://console.groq.com) -2. Obțineți cheia API (nivel gratuit inclus) -3. Tabloul de bord → Adăugați furnizor → Groq +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -**Modele:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Sfat profesionist:**Inferență ultra-rapidă - cel mai bun pentru codare în timp real!### OpenRouter (100+ Models) +### Groq (Free Tier Available!) -1. Înscrieți-vă: [openrouter.ai](https://openrouter.ai) -2. Obțineți cheia API -3. Tabloul de bord → Adăugare furnizor → OpenRouter +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -**Modele:**Accesați peste 100 de modele de la toți furnizorii importanți printr-o singură cheie API. +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Comportamentul tabloului de bord:**modelele OpenRouter sunt gestionate din**Modele disponibile**. Adăugarea manuală, importarea și sincronizarea automată toate actualizează aceeași listă.
+**Pro Tip:** Ultra-fast inference — best for real-time coding! - -💰 Furnizori ieftini (backup)### GLM-4.7 (Daily reset, $0.6/1M) +### OpenRouter (100+ Models) -1. Înscrieți-vă: [Zhipu AI](https://open.bigmodel.cn/) -2. Obțineți cheia API din Coding Plan -3. Tabloul de bord → Adăugați cheia API: - - Furnizor: `glm` - - Cheia API: `cheia-voastra` +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -**Utilizați:**`glm/glm-4.7` +**Models:** Access 100+ models from all major providers through a single API key. -**Sfat profesionist:**Planul de codare oferă cotă de 3 ori la 1/7 cost! Resetați zilnic la 10:00.### MiniMax M2.1 (5h reset, $0.20/1M) +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -1. Înscrieți-vă: [MiniMax](https://www.minimax.io/) -2. Obțineți cheia API -3. Tabloul de bord → Adăugați cheia API + -**Utilizați:**`minimax/MiniMax-M2.1` +
+💰 Cheap Providers (Backup) -**Sfat pro:**Cea mai ieftină opțiune pentru context lung (1 milion de jetoane)!### Kimi K2 ($9/month flat) +### GLM-4.7 (Daily reset, $0.6/1M) -1. Abonați-vă: [Moonshot AI](https://platform.moonshot.ai/) -2. Obțineți cheia API -3. Tabloul de bord → Adăugați cheia API +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**Folosiți:**`kimi/kimi-latest` +**Use:** `glm/glm-4.7` -**Sfat profesionist:**Fix 9 USD/lună pentru 10 milioane de jetoane = 0,90 USD/cost efectiv de 1 milion!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. - -🆓 Furnizori GRATUITI (Backup de urgență)### Qoder (5 FREE models via OAuth) +### MiniMax M2.1 (5h reset, $0.20/1M) + +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `minimax/MiniMax-M2.1` + +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1560,8 +1787,10 @@ Models:
- -🎨 Creați combinații### Example 1: Maximize Subscription → Cheap Backup +
+🎨 Create Combos + +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1589,8 +1818,10 @@ Cost: $0 forever!
- -🔧 Integrare CLI### Cursor IDE +
+🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1601,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Utilizați pagina**Instrumente CLI**din tabloul de bord pentru configurarea cu un singur clic sau editați manual `~/.claude/settings.json`.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1612,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Opțiunea 1 — Tabloul de bord (recomandat):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Opțiunea 2 — Manual:**Editați `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1629,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Notă:**OpenClaw funcționează numai cu OmniRoute local. Utilizați `127.0.0.1` în loc de `localhost` pentru a evita problemele de rezoluție IPv6.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1643,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Pasul 1:**Adăugați OmniRoute ca furnizor personalizat:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Pasul 2:**Creați/editați `opencode.json` în rădăcina proiectului dvs.:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1669,117 +1909,130 @@ opencode } } } -```` +``` -**Pasul 3:**Selectați modelul în OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Sfat:**Adăugați orice model disponibil în punctul final OmniRoute `/v1/models` la secțiunea `modele`. Utilizați formatul `furnizor/model-id` din tabloul de bord OmniRoute.
+ --- ## Depanare - -Dați clic pentru a extinde ghidul de depanare +
+Click to expand troubleshooting guide -**„Modelul de limbă nu a furnizat mesaje”** +**"Language model did not provide messages"** -- Cota de furnizor epuizată → Verificați instrumentul de urmărire a cotei din tabloul de bord -- Soluție: utilizați alternativă combinată sau treceți la un nivel mai ieftin +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**Limitarea ratei** +**Rate limiting** -- Scăderea cotei de abonament → Fallback la GLM/MiniMax -- Adăugați combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**Tokenul OAuth a expirat** +**OAuth token expired** -- Reîmprospătat automat de OmniRoute -- Dacă problemele persistă: Dashboard → Provider → Reconnect +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Costuri mari** +**High costs** -- Verificați statisticile de utilizare în Tabloul de bord → Costuri -- Comutați modelul principal la GLM/MiniMax -- Utilizați nivelul gratuit (Gemini CLI, Qoder) pentru sarcini necritice +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Tabloul de bord/porturile API sunt greșite** +**Dashboard/API ports are wrong** -- `PORT` este portul de bază canonic (și portul API implicit) -- `API_PORT` suprascrie numai ascultatorul API compatibil cu OpenAI -- `DASHBOARD_PORT` suprascrie numai tabloul de bord/ascultătorul Next.js -- Setați `NEXT_PUBLIC_BASE_URL` la tabloul de bord/adresa URL publică (pentru apelurile inverse OAuth) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Erori de sincronizare în cloud** +**Cloud sync errors** -- Verificați punctele `BASE_URL` către instanța dvs. care rulează -- Verificați punctele `CLOUD_URL` către punctul final de cloud așteptat -- Păstrați valorile `NEXT_PUBLIC_*` aliniate cu valorile de pe partea serverului +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Prima conectare nu funcționează** +**First login not working** -- Verificați `INITIAL_PASSWORD` în `.env` -- Dacă nu este setată, parola de rezervă este `123456` +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Fără jurnal de solicitare** +**No request logs** -- Artefactele de solicitare sunt scrise în `DATA_DIR/call_logs/` ca un fișier JSON per solicitare -- Activați capturarea conductei din Dashboard → Jurnale → Solicitați jurnale dacă aveți nevoie de încărcături utile detaliate pe etapă -- Setați `APP_LOG_TO_FILE=true` dacă doriți, de asemenea, jurnalele din consola aplicației în `logs/application/app.log` -- Ajustați `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` și `CALL_LOG_MAX_ENTRIES` după cum este necesar +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Testul de conectare arată „Invalid” pentru furnizorii compatibili cu OpenAI** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Mulți furnizori nu expun un punct final `/models` -- OmniRoute v1.0.6+ include validarea alternativă prin finalizarea chatului -- Asigurați-vă că adresa URL de bază include sufixul „/v1”.### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ Important pentru utilizatorii care rulează OmniRoute pe un VPS, Docker sau orice server la distanță**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -Furnizorii de**Antigravity**și**Gemini CLI**folosesc**Google OAuth 2.0**. Google solicită ca „redirect_uri” din fluxul OAuth să se potrivească exact cu unul dintre URI-urile preînregistrate în Google Cloud Console a aplicației. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -Acreditările OAuth incluse în OmniRoute sunt înregistrate**doar pentru `localhost`**. Când accesați OmniRoute pe un server la distanță (de exemplu, `https://omniroute.myserver.com`), Google respinge autentificarea cu:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Trebuie să creați un**OAuth 2.0 Client ID**în Google Cloud Console cu URI-ul serverului dvs.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Deschide Google Cloud Console** +#### Step-by-step -Accesați: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Creați un nou ID de client OAuth 2.0** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Faceți clic pe**"+ Creați acreditări"**→**"ID client OAuth"** -- Tip aplicație:**"Aplicație web"** -- Nume: orice vă place (de exemplu, „OmniRoute Remote”) +**2. Create a new OAuth 2.0 Client ID** -**3. Adăugați URI de redirecționare autorizate** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -În câmpul**„URI-uri de redirecționare autorizate”**, adăugați:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Înlocuiți `your-server.com` cu domeniul sau IP-ul serverului dvs. (includeți portul dacă este necesar, de exemplu, `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. Salvați și copiați acreditările** +After creating, Google will show the **Client ID** and **Client Secret**. -După creare, Google va afișa**Client ID**și**Client Secret**. +**5. Set environment variables** -**5. Setați variabile de mediu** +In your `.env` (or Docker environment variables): -În `.env` (sau variabilele de mediu Docker):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1788,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Reporniți OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Încercați să vă conectați din nou** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Tabloul de bord → Furnizori → Antigravity (sau Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google va redirecționa acum corect către `https://your-server.com/callback`.--- +--- #### Temporary workaround (without custom credentials) -Dacă nu doriți să vă configurați propriile acreditări chiar acum, puteți utiliza în continuare**fluxul manual de adrese URL**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute deschide adresa URL de autorizare Google -2. După autorizare, Google încearcă să redirecționeze către `localhost` (care nu reușește pe serverul de la distanță) -3.**Copiați adresa URL completă**din bara de adrese a browserului dvs. (chiar dacă pagina nu se încarcă) -4. Lipiți acea adresă URL în câmpul afișat în modalul de conexiune OmniRoute -5. Faceți clic pe**"Conectați"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Acest lucru funcționează deoarece codul de autorizare din URL este valid, indiferent dacă pagina de redirecționare a fost încărcată.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. - -🇧🇷 Versão em Português#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Pentru autentificare,**Antigravity**și**Gemini CLI**folosesc**Google OAuth 2.0**. Google exige că a `redirect_uri` utilizat nu fluxo OAuth seja**exatamente**uma das URIs pre-cadastradas no Google Cloud Console do aplicative. +
+🇧🇷 Versão em Português -În calitate de acreditare OAuth, nu există OmniRoute care sunt cadastrate**apenas pentru `localhost`**. Când accesați OmniRoute într-un server remot (ex: `https://omniroute.meuservidor.com`), sau Google respinge autentificarea com:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Você necesita criar um**OAuth 2.0 Client ID**nu Google Cloud Console ca URI pentru server.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Acces sau Google Cloud Console** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -**2. Plângeți nou OAuth 2.0 Client ID** +**2. Crie um novo OAuth 2.0 Client ID** -- Faceți clic pe**"+ Create Credentials"**→**"OAuth client ID"** -- Tip de aplicație:**"Aplicație web"** -- Nume: scolha qualquer nome (ex: `OmniRoute Remote`) +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Adăugați ca URI de redirecționare autorizate** +**3. Adicione as Authorized Redirect URIs** -No campo**„URI-uri de redirecționare autorizate”**, adiție:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Substitue `seu-servidor.com` pelo domínio sau IP do seu server (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. Salve și copie ca credenciais** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Após criar, o Google afișează o**Client ID**e o**Client Secret**. +**5. Configure as variáveis de ambiente** -**5. Configurați ca variabile de mediu** +No seu `.env` (ou nas variáveis de ambiente do Docker): -Nu ai `.env` (ai variat de ambient do Docker):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1867,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reinicie o OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute +``` -```` +**7. Tente conectar novamente** -**7. Tente connect novamente** +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Tabloul de bord → Furnizori → Antigravity (sau Gemini CLI) → OAuth +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. -Agora sau Google redirecționează corect pentru `https://seu-servidor.com/callback` și funcționează autenticação.--- +--- #### Workaround temporário (sem configurar credenciais próprias) -Nu vă rugăm să vă convingeți acum, dar este posibil să utilizați sau să fluxați**manual de URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. OmniRoute deschide o adresă URL de autorizare Google -2. Após você autorizar, o Google tentará redirecionar for `localhost` (que falha no server remote) -3.**Copiați o adresă URL completă**da bara de accesare a browserului (mesmo que a page não carregue) -4. Cole essa URL nu există câmpuri care nu apar modal de conexão pentru OmniRoute -5. Faceți clic pe**„Conectați-vă”** +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) +4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute +5. Clique em **"Connect"** -> Această soluție de soluționare funcționează deoarece codul de autorizare a URL-ului este valabil independent de redirecționare pentru a încărca sau nu.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1905,64 +2171,73 @@ Nu vă rugăm să vă convingeți acum, dar este posibil să utilizați sau să ## 🛠️ Tech Stack - -Dați clic pentru a extinde detaliile stivei de tehnologie +
+Click to expand tech stack details --**Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ este**nu este acceptat**— binarele native `better-sqlite3` sunt incompatibile) --**Limba**: TypeScript 5.9 —**100% TypeScript**în `src/` și `open-sse/` (zero `orice` în modulele de bază de la v2.0) --**Cadru**: Next.js 16 + React 19 + Tailwind CSS 4 --**Bază de date**: LowDB (JSON) + SQLite (starea domeniului + jurnalele proxy + audit MCP + decizii de rutare) --**Scheme**: Zod (validare I/O instrument MCP, contracte API) --**Protocoale**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Streaming**: evenimente trimise de server (SSE) --**Auth**: OAuth 2.0 (PKCE) + JWT + Chei API + Autorizare MCP --**Testare**: Runner de testare Node.js + Vitest (900+ teste inclusiv unitate, integrare, E2E) --**CI/CD**: GitHub Actions (publicare automată npm + Docker Hub la lansare) --**Site web**: [omniroute.online](https://omniroute.online) --**Pachet**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Reziliență**: întrerupător de circuit, backoff exponențial, turmă anti-tunet, falsificare TLS, auto-vindecare combo
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Documentație -| Document | Descriere | +| Document | Description | | ---------------------------------------------- | --------------------------------------------------- | -| [Ghid de utilizare](docs/USER_GUIDE.md) | Furnizori, combo-uri, integrare CLI, implementare | -| [Referință API](docs/API_REFERENCE.md) | Toate punctele finale cu exemple | -| [Server MCP](open-sse/mcp-server/README.md) | 16 instrumente MCP, configurații IDE, clienți Python/TS/Go | -| [Server A2A](src/lib/a2a/README.md) | Protocol JSON-RPC 2.0, abilități, streaming, gestionarea sarcinilor | -| [Auto-Combo Engine](docs/auto-combo.md) | Scor în 6 factori, pachete de moduri, auto-vindecare | -| [Depanare](docs/TROUBLESHOOTING.md) | Probleme și soluții comune | -| [Arhitectura](docs/ARHITECTURE.md) | Arhitectura sistemului și elementele interne | -| [Contribuind](CONTRIBUTING.md) | Configurare și linii directoare de dezvoltare | -| [OpenAPI Spec](docs/openapi.yaml) | Specificație OpenAPI 3.0 | -| [Politica de securitate](SECURITY.md) | Raportarea vulnerabilităților și practicile de securitate | -| [Implementarea VM](docs/VM_DEPLOYMENT_GUIDE.md) | Ghid complet: VM + nginx + configurare Cloudflare | -| [Galeria de caracteristici](docs/FEATURES.md) | Tur vizual al tabloului de bord cu capturi de ecran | -| [Lista de verificare a lansării](docs/RELEASE_CHECKLIST.md) | Pașii de validare înainte de lansare |--- +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute are**210+ funcții planificate**în mai multe faze de dezvoltare. Iată domeniile cheie: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Categoria | Caracteristici planificate | Repere | -| ----------------------------- | ---------------- | ------------------------------------------------------------------------------------ | -| 🧠**Routing & Intelligence**| 25+ | Rutare cu cea mai mică latență, rutare bazată pe etichete, verificare preliminară a cotei, selecție contului P2C | -| 🔒**Securitate și conformitate**| 20+ | Întărirea SSRF, acoperirea acreditărilor, limita de rată per punct final, domeniul de aplicare al cheii de management | -| 📊**Observabilitate**| 15+ | Integrarea OpenTelemetry, monitorizarea cotelor în timp real, urmărirea costurilor per model | -| 🔄**Integrări furnizori**| 20+ | Registrul modelului dinamic, perioadele de încărcare ale furnizorului, Codexul cu mai multe conturi, analiza cotelor Copilot | -| ⚡**Performanță**| 15+ | Strat cache dublu, cache prompt, cache de răspuns, streaming keepalive, API batch | -| 🌐**Ecosistem**| 10+ | WebSocket API, config hot-reload, magazin de configurare distribuit, mod comercial |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**Integrare OpenCode**— Suport furnizor nativ pentru IDE-ul de codare OpenCode AI -- 🔗**Integrare TRAE**— Suport deplin pentru cadrul de dezvoltare TRAE AI -- 📦**Batch API**— Procesare asincronă în lot pentru solicitări în bloc -- 🎯**Rutare bazată pe etichete**— Solicitări de rutare bazate pe etichete și metadate personalizate -- 💰**Strategia cu cel mai mic cost**— Selectați automat cel mai ieftin furnizor disponibil +### 🔜 Coming Soon -> 📝 Specificații complete disponibile în [`docs/new-features/`](docs/new-features/) (217 specificații detaliate)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1970,18 +2245,20 @@ OmniRoute are**210+ funcții planificate**în mai multe faze de dezvoltare. Iat ### How to Contribute -1. Bifurcați depozitul -2. Creați-vă ramura caracteristică (`git checkout -b feature/amazing-feature`) -3. Commiteți modificările dvs. (`git commit -m 'Adăugați o funcție uimitoare'`) -4. Apăsați la ramură (`caracteristică git push origin/amazing-feature`) -5. Deschideți o cerere de tragere +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Consultați [CONTRIBUTING.md](CONTRIBUTING.md) pentru instrucțiuni detaliate.### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1993,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Mulțumiri speciale pentru**[9router](https://github.com/decolua/9router)**de**[decolua](https://github.com/decolua)**— proiectul original care a inspirat această furcă. OmniRoute se bazează pe această bază incredibilă, cu funcții suplimentare, API-uri multimodale și o rescrie completă TypeScript. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Mulțumiri speciale pentru**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— implementarea Go inițială care a inspirat acest port JavaScript.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Licență -Licență MIT - consultați [LICENȚĂ](LICENȚĂ) pentru detalii.--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/ro/docs/ARCHITECTURE.md b/docs/i18n/ro/docs/ARCHITECTURE.md index 90e3b40a64..4ce36c0e3b 100644 --- a/docs/i18n/ro/docs/ARCHITECTURE.md +++ b/docs/i18n/ro/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Ultima actualizare: 2026-03-28_## Executive Summary -OmniRoute este un gateway local de rutare AI și un tablou de bord construit pe Next.js. -Oferă un singur punct final compatibil cu OpenAI (`/v1/*`) și direcționează traficul către mai mulți furnizori din amonte cu traducere, alternativă, reîmprospătare token și urmărire a utilizării. -Capacitățile de bază: +_Last updated: 2026-03-28_ -- Suprafață API compatibilă cu OpenAI pentru CLI/instrumente (28 de furnizori) -- Traducerea cererii/răspunsurilor între formatele furnizorilor -- Alternativ combo de model (secvență cu mai multe modele) -- Rezervă de rezervă la nivel de cont (cu mai multe conturi pentru fiecare furnizor) -- Gestionarea conexiunii furnizorului OAuth + cheie API -- Generare de încorporare prin `/v1/embeddings` (6 furnizori, 9 modele) -- Generare de imagini prin `/v1/images/generations` (4 furnizori, 9 modele) -- Gândiți-vă la analizarea etichetelor (`...`) pentru modele de raționament -- Sanitizarea răspunsului pentru compatibilitate strictă cu OpenAI SDK -- Normalizarea rolurilor (dezvoltator→sistem, sistem→utilizator) pentru compatibilitate între furnizori -- Conversie de ieșire structurată (json_schema → Gemini responseSchema) -- Persistență locală pentru furnizori, chei, aliasuri, combo-uri, setări, prețuri -- Urmărirea utilizării/costurilor și înregistrarea cererilor -- Sincronizare cloud opțională pentru sincronizare multi-dispozitiv/state -- Lista permisă/lista blocată IP pentru controlul accesului API -- Gândire la managementul bugetului (passthrough/auto/personalizat/adaptativ) -- Sistem global de injectare promptă -- Urmărirea sesiunii și amprentarea -- Limitare îmbunătățită a ratei per cont cu profiluri specifice furnizorului -- Model de întrerupător pentru rezistența furnizorului -- Protectie anti-tunet cu blocare mutex -- Cache de deduplicare a cererilor bazate pe semnătură -- Nivelul domeniului: disponibilitatea modelului, regulile de cost, politica de rezervă, politica de blocare -- Persistența stării domeniului (cache-ul de scriere SQLite pentru rezervări, bugete, blocări, întreruptoare de circuit) -- Motor de politici pentru evaluarea centralizată a cererilor (blocare → buget → rezervă) -- Solicitați telemetrie cu agregarea latenței p50/p95/p99 -- ID de corelare (X-Request-Id) pentru urmărirea de la capăt la capăt -- Înregistrare de audit de conformitate cu renunțare pentru fiecare cheie API -- Cadrul de evaluare pentru asigurarea calității LLM -- Tabloul de bord Resilience UI cu starea întreruptorului în timp real -- Furnizori OAuth modulari (12 module individuale sub `src/lib/oauth/providers/`) +## Executive Summary -Model de rulare principal: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- Rutele aplicației Next.js sub `src/app/api/*` implementează atât API-uri de tablou de bord, cât și API-uri de compatibilitate -- Un nucleu SSE/rutare partajat în `src/sse/*` + `open-sse/*` se ocupă de execuția furnizorului, traducerea, streamingul, fallback-ul și utilizarea## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Timp de rulare gateway local -- API-uri de gestionare a tabloului de bord -- Autentificarea furnizorului și reîmprospătarea simbolului -- Solicitați traducere și streaming SSE -- Stare locală + persistență de utilizare -- Orchestrare opțională de sincronizare în cloud### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Implementarea serviciului cloud în spatele „NEXT_PUBLIC_CLOUD_URL”. -- Furnizor SLA/plan de control în afara procesului local -- Binarele CLI externe în sine (Claude CLI, Codex CLI etc.)## Dashboard Surface (Current) +### Out of Scope -Paginile principale din `src/app/(tabloul de bord)/tabloul de bord/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — pornire rapidă + prezentare generală a furnizorului -- `/dashboard/endpoint` — proxy punct final + MCP + A2A + file API endpoint -- `/dashboard/providers` — conexiuni și acreditări ale furnizorului -- `/dashboard/combos` — strategii combinate, șabloane, reguli de rutare a modelului -- `/dashboard/costs` — agregarea costurilor și vizibilitatea prețurilor -- `/dashboard/analytics` — analize de utilizare și evaluări -- `/dashboard/limits` — controale de cotă/rată -- `/dashboard/cli-tools` — onboarding CLI, detectarea timpului de execuție, generarea config. -- `/dashboard/agents` — agenți ACP detectați + înregistrare personalizată a agentului -- `/dashboard/media` — imagine/video/muzică loc de joacă -- `/dashboard/search-tools` — testarea și istoricul furnizorului de căutare -- `/tableau de bord/sănătate` — timp de funcționare, întrerupătoare, limite de rată -- `/dashboard/logs` — jurnalele cereri/proxy/audit/console -- `/dashboard/settings` — file cu setări de sistem (general, rutare, setări implicite combo etc.) -- `/dashboard/api-manager` — ciclul de viață al cheii API și permisiunile modelului## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Directoare principale: +Main directories: -- `src/app/api/v1/*` și `src/app/api/v1beta/*` pentru API-uri de compatibilitate -- `src/app/api/*` pentru API-uri de gestionare/configurare -- Următoarea rescrie în harta `next.config.mjs` `/v1/*` la `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Rute importante de compatibilitate: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` — include modele personalizate cu `custom: true` -- `src/app/api/v1/embeddings/route.ts` — generare de încorporare (6 furnizori) -- `src/app/api/v1/images/generations/route.ts` — generare de imagini (4+ furnizori inclusiv Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — chat dedicat pentru fiecare furnizor -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — înglobări dedicate pentru fiecare furnizor -- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — imagini dedicate pentru fiecare furnizor +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` -- `src/app/api/v1beta/models/[...cale]/route.ts` +- `src/app/api/v1beta/models/[...path]/route.ts` -Domenii de management: +Management domains: - Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` -- Furnizori/conexiuni: `src/app/api/providers*` -- Noduri furnizor: `src/app/api/provider-nodes*` -- Modele personalizate: `src/app/api/provider-models` (GET/POST/DELETE) -- Catalog de modele: `src/app/api/models/route.ts` (GET) -- Configurare proxy: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- Chei/aliase/combo/preț: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- Utilizare: `src/app/api/usage/*` -- Sincronizare/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` -- Ajutor de instrumente CLI: `src/app/api/cli-tools/*` -- Filtru IP: `src/app/api/settings/ip-filter` (GET/PUT) -- Buget de gândire: `src/app/api/settings/thinking-budget` (GET/PUT) -- prompt de sistem: `src/app/api/settings/system-prompt` (GET/PUT) -- Sesiuni: `src/app/api/sessions` (GET) -- Limite de rată: `src/app/api/rate-limits` (GET) -- Reziliență: `src/app/api/resilience` (GET/PATCH) — profiluri furnizor, întrerupător, stare limită a ratei -- Resetare rezistență: `src/app/api/resilience/reset` (POST) - resetare întrerupătoare + cooldown-uri -- Statistici cache: `src/app/api/cache/stats` (GET/DELETE) -- Disponibilitatea modelului: `src/app/api/models/availability` (GET/POST) -- Telemetrie: `src/app/api/telemetry/summary` (GET) -- Buget: `src/app/api/usage/budget` (GET/POST) -- Lanțuri de rezervă: `src/app/api/fallback/chains` (GET/POST/DELETE) -- Audit de conformitate: `src/app/api/compliance/audit-log` (GET) -- Evaluări: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- Politici: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -Module principale de flux: +## 2) SSE + Translation Core -- Intrare: `src/sse/handlers/chat.ts` -- Orchestrare de bază: `open-sse/handlers/chatCore.ts` -- Adaptoare de execuție furnizor: `open-sse/executors/*` -- Format de detectare/config furnizor: `open-sse/services/provider.ts` -- Analiza/rezolvarea modelului: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Logica de rezervă a contului: `open-sse/services/accountFallback.ts` -- Registrul traducerilor: `open-sse/translator/index.ts` -- Transformări de flux: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- Extragerea/normalizarea utilizării: `open-sse/utils/usageTracking.ts` +Main flow modules: + +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` - Think tag parser: `open-sse/utils/thinkTagParser.ts` - Embedding handler: `open-sse/handlers/embeddings.ts` -- Registrul furnizorului de încorporare: `open-sse/config/embeddingRegistry.ts` -- Manager de generare a imaginii: `open-sse/handlers/imageGeneration.ts` -- Registrul furnizorului de imagini: `open-sse/config/imageRegistry.ts` -- igienizare răspuns: `open-sse/handlers/responseSanitizer.ts` -- Normalizarea rolurilor: `open-sse/services/roleNormalizer.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -Servicii (logica de afaceri): +Services (business logic): -- Selectarea/punctarea contului: `open-sse/services/accountSelector.ts` -- Managementul ciclului de viață context: `open-sse/services/contextManager.ts` -- Aplicarea filtrului IP: `open-sse/services/ipFilter.ts` -- Urmărirea sesiunii: `open-sse/services/sessionManager.ts` -- Solicitați deduplicarea: `open-sse/services/signatureCache.ts` -- Injectarea promptă a sistemului: `open-sse/services/systemPrompt.ts` -- Gândire la managementul bugetului: `open-sse/services/thinkingBudget.ts` -- Rutarea modelului wildcard: `open-sse/services/wildcardRouter.ts` -- Gestionarea limitelor de tarife: `open-sse/services/rateLimitManager.ts` -- Întrerupător: `open-sse/services/circuitBreaker.ts` +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -Module de nivel de domeniu: +Domain layer modules: -- Disponibilitatea modelului: `src/lib/domain/modelAvailability.ts` -- Reguli de cost/bugete: `src/lib/domain/costRules.ts` -- Politica de rezervă: `src/lib/domain/fallbackPolicy.ts` -- Soluție combinată: `src/lib/domain/comboResolver.ts` -- Politica de blocare: `src/lib/domain/lockoutPolicy.ts` -- Motor de politici: `src/domain/policyEngine.ts` — blocare centralizată → buget → evaluare alternativă -- Catalog de coduri de eroare: `src/lib/domain/errorCodes.ts` -- ID cerere: `src/lib/domain/requestId.ts` -- Timeout pentru preluare: `src/lib/domain/fetchTimeout.ts` -- Solicitați telemetrie: `src/lib/domain/requestTelemetry.ts` -- Conformitate/audit: `src/lib/domain/compliance/index.ts` -- Runner de evaluare: `src/lib/domain/evalRunner.ts` -- Persistența stării domeniului: `src/lib/db/domainState.ts` — SQLite CRUD pentru lanțuri de rezervă, bugete, istoricul costurilor, starea de blocare, întrerupătoarele de circuit +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -Module furnizor OAuth (12 fișiere individuale sub `src/lib/oauth/providers/`): +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -- Index de registru: `src/lib/oauth/providers/index.ts` -- Furnizori individuali: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `tski`, `cursor.ts`, `tski`, `locode. -- Ambalaj subțire: `src/lib/oauth/providers.ts` — reexporturi din module individuale## 3) Persistence Layer +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -DB de stat primar (SQLite): +## 3) Persistence Layer + +Primary state DB (SQLite): - Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) -- Reexportați fațada: `src/lib/localDb.ts` (strat subțire de compatibilitate pentru apelanți) -- fișier: `${DATA_DIR}/storage.sqlite` (sau `$XDG_CONFIG_HOME/omniroute/storage.sqlite` când este setat, altfel `~/.omniroute/storage.sqlite`) -- entități (tabele + spații de nume KV): providerConnections, providerNodes, modelAliases, combo, apiKeys, setări, prețuri,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -Persistență de utilizare: +Usage persistence: -- fațadă: `src/lib/usageDb.ts` (module descompuse în `src/lib/usage/*`) -- Tabelele SQLite în `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` -- artefactele de fișier opționale rămân pentru compatibilitate/depanare (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- fișierele JSON moștenite sunt migrate la SQLite prin migrații de pornire atunci când sunt prezente +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -DB Stare Domeniu (SQLite): +Domain State DB (SQLite): -- `src/lib/db/domainState.ts` — operațiuni CRUD pentru starea domeniului -- Tabele (create în `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- Model de cache de scriere: hărțile din memorie sunt autorizate în timpul execuției; mutațiile sunt scrise sincron cu SQLite; starea este restabilită din DB la pornirea la rece## 4) Auth + Security Surfaces +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start -- Autentificare cookie de tablou de bord: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- Generarea/verificarea cheii API: `src/shared/utils/apiKey.ts` -- Secretele furnizorului au persistat în intrările `providerConnections` -- Suport proxy de ieșire prin `open-sse/utils/proxyFetch.ts` (env vars) și `open-sse/utils/networkProxy.ts` (configurabil per furnizor sau global)## 5) Cloud Sync +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync - Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Sarcină periodică: `src/shared/services/cloudSyncScheduler.ts` -- Sarcină periodică: `src/shared/services/modelSyncScheduler.ts` -- Ruta de control: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Deciziile de rezervă sunt conduse de `open-sse/services/accountFallback.ts` folosind coduri de stare și euristica mesajelor de eroare. Rutarea combinată adaugă o protecție suplimentară: 400-urile la nivel de furnizor, cum ar fi eșecurile de blocare a conținutului în amonte și de validare a rolului, sunt tratate ca eșecuri locale ale modelului, astfel încât țintele combo ulterioare să poată rula în continuare.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -Reîmprospătarea în timpul traficului live este executată în `open-sse/handlers/chatCore.ts` prin intermediul executorului `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -Sincronizarea periodică este declanșată de „CloudSyncScheduler” atunci când cloud este activat.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -Fișiere de stocare fizică: +Physical storage files: -- DB primar de rulare: `${DATA_DIR}/storage.sqlite` -- linii de jurnal de solicitare: `${DATA_DIR}/log.txt` (artefact de compatibilitate/depanare) -- arhive structurate de încărcare a apelurilor: `${DATA_DIR}/call_logs/` -- sesiuni opționale de traducător/cerere de depanare: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: API-uri de compatibilitate -- `src/app/api/v1/providers/[provider]/*`: rute dedicate pentru fiecare furnizor (chat, încorporare, imagini) -- `src/app/api/providers*`: furnizor CRUD, validare, testare -- `src/app/api/provider-nodes*`: gestionarea nodurilor compatibile personalizate -- `src/app/api/provider-models`: management personalizat model (CRUD) -- `src/app/api/models/route.ts`: API de catalog de modele (alias-uri + modele personalizate) -- `src/app/api/oauth/*`: fluxuri OAuth/device-code -- `src/app/api/keys*`: ciclul de viață local al cheii API -- `src/app/api/models/alias`: gestionare alias -- `src/app/api/combos*`: gestionarea combo de rezervă -- `src/app/api/pricing`: înlocuirea prețurilor pentru calcularea costurilor -- `src/app/api/settings/proxy`: configurație proxy (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: test de conectivitate proxy de ieșire (POST) -- `src/app/api/usage/*`: API-uri de utilizare și jurnal -- `src/app/api/sync/*` + `src/app/api/cloud/*`: sincronizare în cloud și asistență orientată spre nor -- `src/app/api/cli-tools/*`: scriitori/verificatori de configurare CLI locale -- `src/app/api/settings/ip-filter`: lista IP permisă/lista blocată (GET/PUT) -- `src/app/api/settings/thinking-budget`: configurația bugetului simbolului de gândire (GET/PUT) -- `src/app/api/settings/system-prompt`: prompt de sistem global (GET/PUT) -- `src/app/api/sessions`: listarea sesiunilor active (GET) -- `src/app/api/rate-limits`: starea limitei ratei per cont (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: analizarea cererii, gestionarea combinațiilor, bucla de selecție a contului -- `open-sse/handlers/chatCore.ts`: traducere, expediere executor, reîncercare/reîmprospătare manipulare, configurarea fluxului -- `open-sse/executors/*`: comportamentul de rețea și format specific furnizorului### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: registru și orchestrare a traducătorilor -- Solicitați traducători: `open-sse/translator/request/*` -- Traducători de răspuns: `open-sse/translator/response/*` -- Formatarea constantelor: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: configurație/stare persistentă și persistența domeniului pe SQLite -- `src/lib/localDb.ts`: reexport de compatibilitate pentru modulele DB -- `src/lib/usageDb.ts`: istoricul utilizării/jurnalele de apeluri fațadă deasupra tabelelor SQLite## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Fiecare furnizor are un executor specializat care extinde `BaseExecutor` (în `open-sse/executors/base.ts`), care oferă crearea URL, construcția antetului, reîncercarea cu backoff exponențial, cârlige de reîmprospătare a acreditărilor și metoda de orchestrare `execute()`. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Executant | Furnizor(i) | Manipulare specială | -| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ---------------------------------------------------------------------------------- | -| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Configurare URL dinamică/antet per furnizor | -| `AntigravityExecutor` | Google Antigravity | ID-uri personalizate de proiect/sesiune, Reîncercați-După analizare | -| `CodexExecutor` | OpenAI Codex | Injectează instrucțiuni de sistem, forțează efortul de raționament | -| `CursorExecutor` | Cursor IDE | Protocolul ConnectRPC, codificarea Protobuf, semnarea cererii prin suma de control | -| `GithubExecutor` | GitHub Copilot | Reîmprospătare jeton Copilot, anteturi care imită VSCode | -| `KiroExecutor` | AWS CodeWhisperer/Kiro | Format binar AWS EventStream → conversie SSE | -| `GeminiCLIExecutor` | Gemeni CLI | Ciclul de reîmprospătare a simbolului OAuth Google | +### Persistence -Toți ceilalți furnizori (inclusiv noduri compatibile personalizate) folosesc `DefaultExecutor`.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Furnizor | Format | Auth | Flux | Non-Stream | Token Refresh | Utilizare API | -| ---------------- | ---------------- | ----------------------------- | ---------------- | ---------- | ------------- | ---------------------- | ------------------------------ | -| Claude | claude | Cheie API / OAuth | ✅ | ✅ | ✅ | ⚠️ Doar administrator | -| Gemeni | gemeni | Cheie API / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | -| Gemeni CLI | gemeni-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | -| Antigravitație | antigravitație | OAuth | ✅ | ✅ | ✅ | ✅ Cota completă API | -| OpenAI | deschis | Cheie API | ✅ | ✅ | ❌ | ❌ | -| Codex | openai-responses | OAuth | ✅ forțat | ❌ | ✅ | ✅ Limite de tarif | -| GitHub Copilot | deschis | OAuth + Token Copilot | ✅ | ✅ | ✅ | ✅ Instantanee de cotă | -| Cursor | cursor | Sumă de control personalizată | ✅ | ✅ | ❌ | ❌ | -| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Limite de utilizare | -| Qwen | deschis | OAuth | ✅ | ✅ | ✅ | ⚠️ La cerere | -| Qoder | deschis | OAuth (de bază) | ✅ | ✅ | ✅ | ⚠️ La cerere | -| OpenRouter | deschis | Cheie API | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | claude | Cheie API | ✅ | ✅ | ❌ | ❌ | -| DeepSeek | deschis | Cheie API | ✅ | ✅ | ❌ | ❌ | -| Groq | deschis | Cheie API | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | deschis | Cheie API | ✅ | ✅ | ❌ | ❌ | -| Mistral | deschis | Cheie API | ✅ | ✅ | ❌ | ❌ | -| Nedumerire | deschis | Cheie API | ✅ | ✅ | ❌ | ❌ | -| Împreună AI | deschis | Cheie API | ✅ | ✅ | ❌ | ❌ | -| Artificii AI | deschis | Cheie API | ✅ | ✅ | ❌ | ❌ | -| Cerebre | deschis | Cheie API | ✅ | ✅ | ❌ | ❌ | -| Cohere | deschis | Cheie API | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | deschis | Cheie API | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Formatele sursă detectate includ: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. + +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | + +All other providers (including custom compatible nodes) use the `DefaultExecutor`. + +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: - `openai` - `openai-responses` - `claude` -- `gemeni` +- `gemini` -Formatele țintă includ: +Target formats include: -- Chat/Răspunsuri OpenAI +- OpenAI chat/Responses - Claude -- Plic Gemeni/Gemeni-CLI/Antigravity +- Gemini/Gemini-CLI/Antigravity envelope - Kiro - Cursor -Traducerile folosesc**OpenAI ca format hub**— toate conversiile trec prin OpenAI ca intermediar:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Traducerile sunt selectate dinamic pe baza formei încărcăturii sursei și a formatului țintă al furnizorului. +Additional processing layers in the translation pipeline: -Straturi de procesare suplimentare în conducta de traducere: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Sanitizarea răspunsurilor**— Elimina câmpurile nestandard din răspunsurile în format OpenAI (atât în flux, cât și în non-streaming) pentru a asigura conformitatea strictă cu SDK --**Normalizarea rolurilor**— Convertește `dezvoltator` → `sistem` pentru ținte non-OpenAI; îmbină `sistem` → `utilizator` pentru modelele care resping rolul de sistem (GLM, ERNIE) --**Think tag extraction**— Analizează blocurile `...` din conținut în câmpul `resoning_content` --**Ieșire structurată**— Convertește OpenAI `response_format.json_schema` în `responseMimeType` + `responseSchema` al lui Gemini## Supported API Endpoints +## Supported API Endpoints -| Punct final | Format | Manipulator | -| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------ | -| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | -| `POST /v1/messages` | Claude Mesaje | Același handler (detectat automat) | -| `POST /v1/responses` | Răspunsuri OpenAI | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/embeddings` | Încorporare OpenAI | `open-sse/handlers/embeddings.ts` | -| `GET /v1/embeddings` | Lista de modele | Rută API | -| `POST /v1/images/generations` | Imagini OpenAI | `open-sse/handlers/imageGeneration.ts` | -| `GET /v1/images/generations` | Lista de modele | Rută API | -| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicat pentru fiecare furnizor cu validare a modelului | -| `POST /v1/providers/{provider}/embeddings` | Încorporare OpenAI | Dedicat pentru fiecare furnizor cu validare a modelului | -| `POST /v1/providers/{provider}/images/generations` | Imagini OpenAI | Dedicat pentru fiecare furnizor cu validare a modelului | -| `POST /v1/messages/count_tokens` | Claude Token Count | Rută API | -| `GET /v1/models` | Lista de modele OpenAI | Rută API (chat + încorporare + imagine + modele personalizate) | -| `GET /api/models/catalog` | Catalog | Toate modelele grupate după furnizor + tip | -| `POST /v1beta/models/*:streamGenerateContent` | nativ Gemeni | Rută API | -| `GET/PUT/DELETE /api/settings/proxy` | Configurare proxy | Configurare proxy de rețea | -| `POST /api/settings/proxy/test` | Conectivitate proxy | Punct final de testare de sănătate/conectivitate proxy | -| `GET/POST/DELETE /api/provider-models` | Modele de furnizori | Metadatele modelului furnizorului care susțin modelele disponibile personalizate și gestionate |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -Managerul de ocolire (`open-sse/utils/bypassHandler.ts`) interceptează cererile cunoscute „de aruncat” de la Claude CLI — ping-uri de încălzire, extrageri de titluri și numărătoare de jetonuri — și returnează un**răspuns fals**fără a consuma jetoane de furnizor în amonte. Acest lucru este declanșat numai când `User-Agent` conține `claude-cli`.## Request Logger Pipeline +## Bypass Handler -Loggerul de solicitare (`open-sse/utils/requestLogger.ts`) oferă o conductă de înregistrare a depanării în 7 etape, dezactivată implicit, activată prin `ENABLE_REQUEST_LOGS=true`:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Fișierele sunt scrise în `/logs//` pentru fiecare sesiune de solicitare.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- cooldown contului furnizorului pentru erori tranzitorii/rate/auth -- rezervă de cont înainte de cererea eșuată -- alternativă model combo atunci când modelul curent/calea furnizorului este epuizată## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- preverificare și reîmprospătare cu reîncercare pentru furnizorii care pot fi reîmprospătați -- 401/403 reîncercați după încercarea de reîmprospătare în calea de bază## 3) Stream Safety +## 2) Token Expiry -- controler de flux conștient de deconectare -- flux de traducere cu spălare la sfârșitul fluxului și gestionarea `[DONE]` -- estimarea utilizării de rezervă atunci când metadatele de utilizare ale furnizorului lipsesc## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- apar erori de sincronizare, dar timpul de execuție local continuă -- planificatorul are o logică capabilă să reîncerce, dar execuția periodică apelează în mod implicit sincronizarea cu o singură încercare## 5) Data Integrity +## 3) Stream Safety -- Migrații de schemă SQLite și cârlige de actualizare automată la pornire -- moștenire JSON → cale de compatibilitate cu migrarea SQLite## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Surse de vizibilitate la runtime: +## 4) Cloud Sync Degradation -- jurnalele consolei de la `src/sse/utils/logger.ts` -- agregate de utilizare pe cerere în SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- capturi detaliate de încărcare utilă în patru etape în SQLite (`request_detail_logs`) când `settings.detailed_logs_enabled=true` -- Jurnalul de stare a cererii textuale în `log.txt` (opțional/compat) -- jurnalele opționale de solicitare/traducere profundă sub `jurnale/` când `ENABLE_REQUEST_LOGS=true` -- puncte finale de utilizare a tabloului de bord (`/api/usage/*`) pentru consumul UI +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -Captura detaliată a sarcinii utile a cererii stochează până la patru etape JSON de încărcare utilă pentru fiecare apel direcționat: +## 5) Data Integrity -- cerere bruta primita de la client -- cerere tradusă trimisă efectiv în amonte -- răspunsul furnizorului reconstruit ca JSON; răspunsurile transmise în flux sunt compactate în rezumatul final plus metadatele fluxului -- răspunsul final al clientului returnat de OmniRoute; răspunsurile transmise în flux sunt stocate în aceeași formă de rezumat compact## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- Secretul JWT (`JWT_SECRET`) securizează verificarea/semnarea cookie-urilor sesiunii de bord -- Bootstrap-ul inițial al parolei (`INITIAL_PASSWORD`) ar trebui să fie configurat în mod explicit pentru furnizarea la prima executare -- Cheia API secretă HMAC (`API_KEY_SECRET`) securizează formatul cheii API locale generate -- Secretele furnizorului (chei/token-uri API) sunt păstrate în DB local și ar trebui protejate la nivel de sistem de fișiere -- Punctele finale de sincronizare în cloud se bazează pe semantica de autentificare a cheii API + ID-ul mașinii## Environment and Runtime Matrix +## Observability and Operational Signals -Variabilele de mediu utilizate în mod activ de cod: +Runtime visibility sources: -- Aplicație/autentificare: `JWT_SECRET`, `INITIAL_PASSWORD` -- Stocare: `DATA_DIR` -- Comportamentul nodului compatibil: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Suprascrierea opțională a bazei de stocare (Linux/macOS când `DATA_DIR` este dezactivat): `XDG_CONFIG_HOME` -- Hashing de securitate: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Înregistrare: `ENABLE_REQUEST_LOGS` -- URL sincronizare/cloud: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- Proxy de ieșire: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` și variante cu litere mici -- Indicatori de caracteristică SOCKS5: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- Ajutor platformă/execuție (configurație nu specifică aplicației): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` și `localDb` au aceeași politică de bază de director (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) cu migrarea fișierelor moștenite. -2. `/api/v1/route.ts` se deleagă la același constructor de catalog unificat folosit de `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) pentru a evita deriva semantică. -3. Loggerul solicitărilor scrie anteturi/corp complet atunci când este activat; tratați directorul de jurnal ca fiind sensibil. -4. Comportamentul în cloud depinde de „NEXT_PUBLIC_BASE_URL” corect și de accesibilitatea punctului final din cloud. -5. Directorul `open-sse/` este publicat ca pachetul de spațiu de lucru `@omniroute/open-sse`**npm**. Codul sursă îl importă prin `@omniroute/open-sse/...` (rezolvat de Next.js `transpilePackages`). Căile fișierelor din acest document încă folosesc numele de director `open-sse/` pentru consecvență. -6. Diagramele din tabloul de bord utilizează**Recharts**(bazate pe SVG) pentru vizualizări analitice accesibile, interactive (diagrame cu bare de utilizare a modelelor, tabele de defalcare a furnizorilor cu rate de succes). -7. Testele E2E folosesc**Playwright**(`tests/e2e/`), rulează prin `npm run test:e2e`. Testele unitare folosesc**Node.js test runner**(`tests/unit/`), rulează prin `npm run test:unit`. Codul sursă sub `src/` este**TypeScript**(`.ts`/`.tsx`); spațiul de lucru `open-sse/` rămâne JavaScript (`.js`). -8. Pagina Setări este organizată în 5 file: Securitate, Rutare (6 strategii globale: fill-first, round-robin, p2c, aleatoriu, cel mai puțin utilizat, optimizat pentru cost), Reziliență (limite ale ratei editabile, întrerupător de circuit, politici), AI (buget de gândire, prompt de sistem, cache prompt), Avansat (proxy).## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- Construire din sursă: `npm run build` -- Build Docker imagine: `docker build -t omniroute .` -- Porniți serviciul și verificați: +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: - `GET /api/settings` - `GET /api/v1/models` -- Adresa URL de bază țintă CLI ar trebui să fie `http://:20128/v1` când `PORT=20128` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/ro/docs/FEATURES.md b/docs/i18n/ro/docs/FEATURES.md index 315556582b..66d92a4fa1 100644 --- a/docs/i18n/ro/docs/FEATURES.md +++ b/docs/i18n/ro/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Ghid vizual pentru fiecare secțiune a tabloului de bord OmniRoute.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Gestionați conexiunile furnizorilor AI: furnizori OAuth (Claude Code, Codex, Gemini CLI), furnizori de chei API (Groq, DeepSeek, OpenRouter) și furnizori gratuiti (Qoder, Qwen, Kiro). Conturile Kiro includ urmărirea soldului creditului — creditele rămase, alocația totală și data de reînnoire sunt vizibile în Tabloul de bord → Utilizare.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Creați combinații de modele de rutare cu 6 strategii: prioritar, ponderat, round-robin, aleatoriu, cel mai puțin utilizat și optimizat din punct de vedere al costurilor. Fiecare combo înlănțuiește mai multe modele cu rezervă automată și include șabloane rapide și verificări de pregătire.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Analiză cuprinzătoare a utilizării cu consum de simboluri, estimări de costuri, hărți termice ale activității, diagrame de distribuție săptămânală și defalcări pentru fiecare furnizor.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Monitorizare în timp real: timp de funcționare, memorie, versiune, percentile de latență (p50/p95/p99), statistici cache și stări întrerupătoarelor furnizorului.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Patru moduri de depanare a traducerilor API:**Playground**(convertor de format),**Chat Tester**(cereri live),**Test Bench**(testare în lot) și**Live Monitor**(stream în timp real).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Testați orice model direct de pe tabloul de bord. Selectați furnizorul, modelul și punctul final, scrieți solicitări cu Editorul Monaco, transmiteți răspunsuri în timp real, anulați fluxul la mijloc și vizualizați valorile de sincronizare.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Teme de culoare personalizabile pentru întreg tabloul de bord. Alegeți dintre cele 7 culori prestabilite (Coral, Albastru, Roșu, Verde, Violet, Portocaliu, Cyan) sau creați o temă personalizată alegând orice culoare hexagonală. Acceptă modul de lumină, întuneric și sistem.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Panou cuprinzător de setări cu file: +Comprehensive settings panel with tabs: --**General**— Stocare de sistem, management de backup (bază de date de export/import) -**Aspect**— Selector de teme (întuneric/luminos/sistem), presetări pentru teme de culoare și culori personalizate, vizibilitate jurnal de sănătate, comenzi pentru vizibilitatea elementelor din bara laterală -**Securitate**— protecție API finală, blocare personalizată a furnizorului, filtrare IP, informații despre sesiune -**Routing**— Aliasuri de model, degradarea sarcinilor de fundal -**Reziliență**— Persistența limitei ratei, reglarea întrerupătorului, dezactivarea automată a conturilor interzise, monitorizarea expirării furnizorului -**Avansat**— Modificari de configurare, urmărire de auditare a configurației, modul de degradare de rezervă![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Configurare cu un singur clic pentru instrumente de codare AI: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor și Factory Droid. Dispune de aplicare/resetare automată a configurației, profiluri de conexiune și mapare a modelului.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Tabloul de bord pentru descoperirea și gestionarea agenților CLI. Afișează o grilă de 14 agenți încorporați (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) cu: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Stare de instalare**— Instalat/Negăsit cu detectarea versiunii -**Insigne de protocol**— stdio, HTTP etc. -**Agenți personalizați**- Înregistrați orice instrument CLI prin formular (nume, binar, comandă de versiune, spawn args) -**Potrivirea amprentei CLI**— Comută pe furnizor pentru a potrivi semnăturile de solicitare CLI native, reducând riscul de interzicere, păstrând IP-ul proxy--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Generați imagini, videoclipuri și muzică din tabloul de bord. Suportă OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open și MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Înregistrare în timp real a cererilor cu filtrare în funcție de furnizor, model, cont și cheie API. Afișează codurile de stare, utilizarea simbolurilor, latența și detaliile răspunsului.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Punctul final API unificat cu defalcarea capacităților: Finalizări de chat, API de răspunsuri, încorporare, generare de imagini, reclasificare, transcriere audio, text-to-speech, moderări și chei API înregistrate. Integrare Cloudflare Quick Tunnel și suport proxy cloud pentru acces de la distanță.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -Creați, acoperiți și revocați cheile API. Fiecare cheie poate fi restricționată la anumite modele/furnizori cu acces complet sau permisiuni numai pentru citire. Gestionarea vizuală a cheilor cu urmărirea utilizării.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Urmărirea acțiunilor administrative cu filtrare după tip de acțiune, actor, țintă, adresă IP și marcaj de timp. Istoricul complet al evenimentelor de securitate.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Aplicația desktop nativă Electron pentru Windows, macOS și Linux. Rulați OmniRoute ca aplicație autonomă cu integrare în bara de sistem, asistență offline, actualizare automată și instalare cu un singur clic. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Caracteristici cheie: +Key features: -- Sondaj de pregătire a serverului (fără ecran gol la pornirea la rece) -- Tava de sistem cu management port -- Politica de securitate a conținutului -- Blocare cu o singură instanță -- Actualizare automată la repornire -- Interfață de utilizare condiționată de platformă (semafor macOS, bara de titlu implicită Windows/Linux) -- Hardened Electron build package — `node_modules` cu legături simbolice din pachetul autonom este detectat și respins înainte de împachetare, prevenind dependența de rulare de mașina de construire (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Consultați [`electron/README.md`](../electron/README.md) pentru documentația completă. +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/ro/docs/TROUBLESHOOTING.md b/docs/i18n/ro/docs/TROUBLESHOOTING.md index 33d445baee..e603e9c1e6 100644 --- a/docs/i18n/ro/docs/TROUBLESHOOTING.md +++ b/docs/i18n/ro/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -Probleme și soluții comune pentru OmniRoute.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Problemă | Soluție | -| -------------------------------------------- | ----------------------------------------------------------------------------- | --- | -| Prima conectare nu funcționează | Setați `INITIAL_PASSWORD` în `.env` (fără cod implicit implicit) | -| Tabloul de bord se deschide pe portul greșit | Setați `PORT=20128` și `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Niciun jurnal de solicitare sub `logs/` | Setați `ENABLE_REQUEST_LOGS=true` | -| EACCES: permisiunea refuzată | Setați `DATA_DIR=/path/to/writable/dir` pentru a înlocui `~/.omniroute` | -| Strategia de rutare nu se salvează | Actualizare la v1.4.11+ (remedierea schemei Zod pentru persistența setărilor) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Cauza:**Cota de furnizor a fost epuizată. +**Cause:** Provider quota exhausted. -**Remediere:** +**Fix:** -1. Verificați instrumentul de urmărire a cotelor din tabloul de bord -2. Utilizați un combo cu niveluri de rezervă -3. Treceți la nivelul mai ieftin/gratuit### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Cauza:**Cota de abonament epuizată. +### Rate Limiting -**Remediere:** +**Cause:** Subscription quota exhausted. -- Adăugați alternativă: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Utilizați GLM/MiniMax ca rezervă ieftină### OAuth Token Expired +**Fix:** -OmniRoute reîmprospătează automat jetoanele. Dacă problemele persistă: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Tabloul de bord → Furnizor → Reconectare -2. Ștergeți și adăugați din nou conexiunea la furnizor--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Verificați `BASE_URL` punctele către instanța dvs. care rulează (de ex., `http://localhost:20128`) -2. Verificați `CLOUD_URL` punctele către punctul final de cloud (de exemplu, `https://omniroute.dev`) -3. Păstrați valorile `NEXT_PUBLIC_*` aliniate cu valorile de pe partea serverului### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Simptom:**`Token neașteptat 'd'...` pe punctul final de cloud pentru apeluri care nu sunt transmise în flux. +### Cloud `stream=false` Returns 500 -**Cauza:**Upstream returnează sarcina utilă SSE în timp ce clientul așteaptă JSON. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Soluție:**Folosiți `stream=true` pentru apelurile directe în cloud. Timpul de rulare local include SSE→JSON fallback.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Creați o cheie nouă din tabloul de bord local (`/api/keys`) -2. Rulați sincronizarea în cloud: Activați Cloud → Sincronizare acum -3. Cheile vechi/nesincronizate pot returna în continuare `401` pe cloud--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Verificați câmpurile runtime: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. Pentru modul portabil: utilizați imaginea țintă „runner-cli” (CLI-uri incluse) -3. Pentru modul de montare a gazdei: setați `CLI_EXTRA_PATHS` și montați directorul bin gazdă ca doar pentru citire -4. Dacă `installed=true` și `runnable=false`: binarul a fost găsit, dar verificarea de sănătate a eșuat### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Verificați statisticile de utilizare în Tabloul de bord → Utilizare -2. Comutați modelul principal la GLM/MiniMax -3. Utilizați nivelul gratuit (Gemini CLI, Qoder) pentru sarcini non-critice -4. Setați bugete de cost pentru fiecare cheie API: Tabloul de bord → Chei API → Buget--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Setați `ENABLE_REQUEST_LOGS=true` în fișierul dvs. `.env`. Jurnalele apar în directorul `logs/`.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Stare principală: `${DATA_DIR}/storage.sqlite` (furnizori, combo-uri, aliasuri, chei, setări) -- Utilizare: tabele SQLite în `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + opțional `${DATA_DIR}/log.txt` și `${DATA_DIR}/call_logs/` -- Jurnalele de solicitare: `/logs/...` (când `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Când întrerupătorul unui furnizor este DESCHIS, cererile sunt blocate până la expirarea perioadei de răcire. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Remediere:** +**Fix:** -1. Accesați**Tabloul de bord → Setări → Reziliență** -2. Verificați cardul întreruptorului pentru furnizorul afectat -3. Faceți clic pe**Reset All**pentru a șterge toate întrerupătoarele sau așteptați ca perioada de răcire să expire -4. Verificați că furnizorul este efectiv disponibil înainte de resetare### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Dacă un furnizor intră în mod repetat în starea DESCHIS: +### Provider keeps tripping the circuit breaker -1. Verificați**Tabloul de bord → Sănătate → Sănătatea furnizorului**pentru modelul de eșec -2. Accesați**Setări → Reziliență → Profiluri furnizor**și creșteți pragul de eșec -3. Verificați dacă furnizorul a modificat limitele API sau dacă necesită re-autentificare -4. Examinați telemetria latenței — latența mare poate cauza eșecuri bazate pe timeout--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Asigurați-vă că utilizați prefixul corect: `deepgram/nova-3` sau `assemblyai/best` -- Verificați că furnizorul este conectat în**Tabloul de bord → Furnizori**### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Verificați formatele audio acceptate: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- Verificați că dimensiunea fișierului este în limitele furnizorului (de obicei < 25 MB) -- Verificați valabilitatea cheii API a furnizorului în cardul furnizorului--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Utilizați**Tabloul de bord → Traducător**pentru a depana problemele de traducere de format: +Use **Dashboard → Translator** to debug format translation issues: -| Modul | Când să utilizați | -| ------------------- | ----------------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Teren de joacă** | Comparați formatele de intrare/ieșire una lângă alta — inserați o solicitare eșuată pentru a vedea cum se traduce | -| **Tester de chat** | Trimiteți mesaje live și inspectați întreaga sarcină de solicitare/răspuns, inclusiv antetele | -| **Banc de testare** | Rulați teste în loturi în combinații de formate pentru a afla ce traduceri sunt întrerupte | -| **Monitor live** | Urmăriți fluxul de solicitări în timp real pentru a detecta problemele intermitente de traducere | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Nu apar etichete de gândire**— Verificați dacă furnizorul țintă acceptă gândirea și setarea bugetului de gândire -**Scăderea apelurilor de instrumente**— Unele traduceri în format pot elimina câmpurile neacceptate; verificați în modul Playground -**Lipsește promptul de sistem**— Claude și Gemini gestionează prompturile în mod diferit; verificați rezultatul traducerii -**SDK returnează șir brut în loc de obiect**— Remediat în v1.1.0: dezinfectantul de răspuns acum elimină câmpurile nestandard (`x_groq`, `usage_breakdown`, etc.) care cauzează eșecuri de validare OpenAI SDK Pydantic -**GLM/ERNIE respinge rolul „sistemului”**— Remediat în v1.1.0: normalizatorul de rol îmbină automat mesajele de sistem în mesajele utilizatorului pentru modele incompatibile -**Rolul `dezvoltator` nu este recunoscut**— Remediat în v1.1.0: convertit automat în `sistem` pentru furnizorii non-OpenAI -**`json_schema` nu funcționează cu Gemini**— Remediat în v1.1.0: `response_format` este acum convertit în `responseMimeType` + `responseSchema`--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- Limita automată a ratei se aplică numai furnizorilor de chei API (nu OAuth/abonament) -- Verificați că**Setări → Reziliență → Profiluri furnizorului**are limita de rata automată activată -- Verificați dacă furnizorul returnează codurile de stare `429` sau anteturile `Retry-After`### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -Profilurile furnizorilor acceptă aceste setări: +### Tuning exponential backoff --**Întârziere de bază**— Timp de așteptare inițial după prima defecțiune (implicit: 1s) -**Întârziere maximă**— Limită maximă a timpului de așteptare (implicit: 30s) -**Multiplicator**— Cât de mult se mărește întârzierea pentru fiecare defecțiune consecutivă (implicit: 2x)### Anti-thundering herd +Provider profiles support these settings: -Când multe solicitări concurente ajung la un furnizor cu o rată limitată, OmniRoute folosește mutex + limitarea automată a ratei pentru a serializa cererile și a preveni eșecurile în cascadă. Acest lucru este automat pentru furnizorii de chei API.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Unii utilizatori OmniRoute plasează gateway-ul în fața RAG sau a stivelor de agenți. În acele setări este obișnuit să vezi un model ciudat: OmniRoute arată sănătos (furnizorii sus, profiluri de rutare ok, alerte fără limită de rată), dar răspunsul final este încă greșit. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -În practică, aceste incidente provin de obicei de la conducta RAG din aval, nu de la gateway în sine. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Dacă doriți un vocabular comun pentru a descrie acele defecțiuni, puteți utiliza WFGY ProblemMap, o resursă text externă a licenței MIT care definește șaisprezece modele de eșec RAG / LLM recurente. La un nivel înalt acoperă: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- deriva de regăsire și granițele de context rupte -- indexuri goale sau învechite și depozite de vectori -- încorporare versus nepotrivire semantică -- probleme de asamblare promptă și ferestre de context -- colaps logic și răspunsuri prea încrezătoare -- eșecuri de coordonare a lanțului lung și a agenților -- memorie multi-agent și deriva de rol -- probleme de implementare și comanda bootstrap +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -Ideea este simpla: +The idea is simple: -1. Când investigați un răspuns prost, capturați: - - sarcina și cererea utilizatorului - - combo rută sau furnizor în OmniRoute - - orice context RAG utilizat în aval (documente preluate, apeluri de instrumente etc.) -2. Harta incidentul la unul sau două numere WFGY ProblemMap (`Nr.1` … `Nr.16`). -3. Stocați numărul în propriul tablou de bord, runbook sau instrument de urmărire a incidentelor lângă jurnalele OmniRoute. -4. Utilizați pagina WFGY corespunzătoare pentru a decide dacă trebuie să vă schimbați stiva RAG, retriever sau strategia de rutare. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -Text complet și rețete concrete live aici (licență MIT, doar text): +Full text and concrete recipes live here (MIT license, text only): [WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -Puteți ignora această secțiune dacă nu rulați RAG sau conducte de agenți în spatele OmniRoute.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**Probleme GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Arhitectura**: Vezi [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) pentru detalii interne -**Referință API**: Consultați [`docs/API_REFERENCE.md`](API_REFERENCE.md) pentru toate punctele finale -**Tabloul de bord pentru sănătate**: verificați**Tabloul de bord → Sănătate**pentru starea sistemului în timp real -**Translator**: utilizați**Tabloul de bord → Translator**pentru a depana problemele de format +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/ro/llm.txt b/docs/i18n/ro/llm.txt new file mode 100644 index 0000000000..d8413e3d70 --- /dev/null +++ b/docs/i18n/ro/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Română) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Prezentare generală + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Securitate +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/ru/README.md b/docs/i18n/ru/README.md index 7e65f03dd0..3fac1f2779 100644 --- a/docs/i18n/ru/README.md +++ b/docs/i18n/ru/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Ваш универсальный API-прокси — одна конечная точка, более 60 провайдеров, нулевое время простоя. Теперь с**MCP-сервером (25 инструментов)**,**Протоколом A2A**,**Системами памяти/навыков**и**Приложением Electronic Desktop**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Завершение чата • Встраивание • Генерация изображений • Видео • Музыка • Аудио • Изменение рейтинга •**Веб-поиск**• Сервер MCP • Протокол A2A • 100 % TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Ваш универсальный API-прокси — одна конечна [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Веб-сайт](https://omniroute.online) • [🚀 Быстрый старт](#-быстрый старт) • [💡 Возможности](#-ключевые функции) • [📖 Документация](#-документация) • [💰 Цены](#-цены-краткий обзор) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Доступно на:**🇺🇸 [английский](README.md) | 🇧🇷 [Португальский (Бразилия)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Испанский](docs/i18n/es/README.md) | 🇫🇷 [Французский](docs/i18n/fr/README.md) | 🇮🇹 [Итальянский](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Суоми](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Мадьяр](docs/i18n/hu/README.md) | 🇮🇩 [Бахаса Индонезия](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Бахаса Мелаю](docs/i18n/ms/README.md) | 🇳🇱 [Нидерланды](docs/i18n/nl/README.md) | 🇳🇴 [Норск](docs/i18n/no/README.md) | 🇵🇹 [Português (Португалия)](docs/i18n/pt/README.md) | 🇷🇴 [Романэ](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Словенчина](docs/i18n/sk/README.md) | 🇸🇪 [Свенска](docs/i18n/sv/README.md) | 🇵🇭 [Филиппинский](docs/i18n/phi/README.md) | 🇨🇿 [Чештина](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,554 +60,629 @@ _Ваш универсальный API-прокси — одна конечна ## 📸 Dashboard Preview -<подробности> +
+Click to see dashboard screenshots -Нажмите, чтобы просмотреть снимки экрана панели управления +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | -| Страница | Скриншот | -| ------------------------------------------- | ----------------------------------------------------- | ---------- | -| **Поставщики** | ![Поставщики](docs/screenshots/01-providers.png) | -| **Комбинации** | ![Комбо](docs/screenshots/02-combos.png) | -| **Аналитика** | ![Аналитика](docs/screenshots/03-analytics.png) | -| **Здоровье** | ![Здоровье](docs/screenshots/04-health.png) | -| **Переводчик** | ![Переводчик](docs/screenshots/05-translator.png) | -| **Настройки** | ![Настройки](docs/screenshots/06-settings.png) | -| **Инструменты интерфейса командной строки** | ![Инструменты CLI](docs/screenshots/07-cli-tools.png) | -| **Журналы использования** | ![Использование](docs/screenshots/08-usage.png) | -| **Конечные точки** | ![Конечные точки](docs/screenshots/09-endpoint.png) |
| + --- ### 🤖 Free AI Provider for your favorite coding agents -_Подключите любую среду IDE или интерфейс командной строки на базе искусственного интеллекта через OmniRoute — бесплатный шлюз API для неограниченного кодирования._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ -<таблица> -<тр> - - -OpenClaw
-ОпенКло -

-⭐ 205 тыс. - - - -NanoBot
-НаноБот -

-⭐ 20,9 тыс. - - - -PicoClaw
-Пикокоготь -

-⭐ 14,6 тыс. - - - -ZeroClaw
-Нулевой коготь -

-⭐ 9,9 тыс. - - - -IronClaw
-Железный коготь -

-⭐ 2,1 тыс. - - -<тр> - - -OpenCode
-Открытый код -

-⭐ 106 тыс. - - - -CLI Codex
-Интерфейс командной строки Кодекса -

-⭐ 60,8 тыс. - - - -Код Клода
-Код Клауда -

-⭐ 67,3 тыс. - - - -Gemini CLI
-Gemini CLI -

-⭐ 94,7 тыс. - - - -Код Kilo
-Код килограмма -

-⭐ 15,5 тыс. - - - + + + + + + + + + + + + + + + +
+ + OpenClaw
+ OpenClaw +

+ ⭐ 205K +
+ + NanoBot
+ NanoBot +

+ ⭐ 20.9K +
+ + PicoClaw
+ PicoClaw +

+ ⭐ 14.6K +
+ + ZeroClaw
+ ZeroClaw +

+ ⭐ 9.9K +
+ + IronClaw
+ IronClaw +

+ ⭐ 2.1K +
+ + OpenCode
+ OpenCode +

+ ⭐ 106K +
+ + Codex CLI
+ Codex CLI +

+ ⭐ 60.8K +
+ + Claude Code
+ Claude Code +

+ ⭐ 67.3K +
+ + Gemini CLI
+ Gemini CLI +

+ ⭐ 94.7K +
+ + Kilo Code
+ Kilo Code +

+ ⭐ 15.5K +
-📡 Все агенты подключаются через http://localhost:20128/v1 или http://cloud.omniroute.online/v1 — одна конфигурация, неограниченное количество моделей и квота--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Хватит тратить деньги и превышать лимиты:** +**Stop wasting money and hitting limits:** -- Квота подписки истекает каждый месяц, если она не используется. -- Ограничения скорости не позволяют вам кодировать в середине -- Дорогие API (20–50 долларов США в месяц на каждого поставщика) -- Ручное переключение между провайдерами +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute решает эту проблему:** +**OmniRoute solves this:** -- ✅**Максимальное количество подписок**- Отслеживайте квоту, используйте каждый бит перед сбросом -- ✅**Автоматический возврат**- Подписка → Ключ API → Дешево → Бесплатно, нулевое время простоя -- ✅**Мультиаккаунт**— Циклическое переключение между аккаунтами каждого провайдера. -- ✅**Универсальность**— Работает с Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw и любым инструментом CLI.--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Присоединяйтесь к нашему сообществу!**[Группа WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Получайте помощь, делитесь советами и будьте в курсе событий. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Веб-сайт**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Проблемы**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Группа сообщества](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Вклад**: посмотрите [CONTRIBUTING.md](CONTRIBUTING.md), откройте PR или выберите «хороший первый выпуск». -**Оригинальный проект**: [9router от decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -При открытии задачи выполните команду system-info и прикрепите сгенерированный файл:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -При этом создается файл system-info.txt с вашей версией Node.js, версией OmniRoute, сведениями об ОС, установленными инструментами CLI (qoder, Gemini, Claude, Codex, AntiGravity, Droid и т. д.), статусом Docker/PM2 и системными пакетами — всем, что нам нужно для быстрого воспроизведения вашей проблемы. Прикрепите файл непосредственно к вашей задаче на GitHub.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Каждый разработчик, использующий инструменты искусственного интеллекта, ежедневно сталкивается с этими проблемами.**OmniRoute был создан для решения всех этих проблем — от перерасхода средств до региональных блоков, от нарушенных потоков OAuth до операций протокола и наблюдения за предприятием. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. -<подробности> -💸 1. «Я плачу за дорогую подписку, но меня все равно прерывают лимиты» +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Разработчики платят 20–200 долларов в месяц за Claude Pro, Codex Pro или GitHub Copilot. Даже при оплате квота имеет потолок — 5 часов использования, еженедельные лимиты или поминутные ограничения. В середине сеанса кодирования провайдер перестает отвечать, и разработчик теряет поток и производительность. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Как OmniRoute решает эту проблему:** +**How OmniRoute solves it:** --**Умный 4-уровневый резерв**— если квота подписки исчерпана, происходит автоматическое перенаправление на API-ключ → Дешево → Бесплатно без вмешательства вручную. --**Отслеживание ограничений поставщика** — снимки кэшированных квот обновляются по расписанию на стороне сервера (по умолчанию `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) с ручным обновлением, доступным в пользовательском интерфейсе. --**Поддержка нескольких учетных записей**— Несколько учетных записей у каждого провайдера с автоматическим циклическим перебором — когда один из них заканчивается, переключается на следующий --**Пользовательские комбинации**— Настраиваемые резервные цепочки с 9 стратегиями балансировки (приоритет, взвешенная, сначала заполняется, циклический, P2C, случайный, наименее используемый, оптимизированный по затратам, строго случайный) --**Бизнес-квоты Кодекса**— мониторинг квот рабочего пространства для бизнеса/команды непосредственно на панели управления.
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard -<подробности> -🔌 2. «Мне нужно использовать несколько поставщиков, но у каждого свой API» + -OpenAI использует один формат, Claude (Anthropic) — другой, Gemini — третий. Если разработчик хочет протестировать модели от разных поставщиков или использовать резервный вариант между ними, ему необходимо перенастроить SDK, изменить конечные точки, разобраться с несовместимыми форматами. Пользовательские поставщики (FriendLI, NIM) имеют нестандартные конечные точки модели. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Как OmniRoute решает эту проблему:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Единая конечная точка** — один `http://localhost:20128/v1` служит прокси для всех более чем 60 провайдеров. --**Перевод формата**— Автоматический и прозрачный: OpenAI ↔ Claude ↔ Gemini ↔ API ответов --**Очистка ответов**— удаляются нестандартные поля (`x_groq`, `usage_breakdown`, `service_tier`), которые нарушают работу OpenAI SDK v1.83+. --**Нормализация ролей**— преобразует «разработчик» → «система» для поставщиков, не поддерживающих OpenAI; `система` → `пользователь` для GLM/ERNIE --**Think Tag Extraction**— извлекает блоки `` из таких моделей, как DeepSeek R1, в стандартизированный `reasoning_content`. --**Структурированный вывод для Gemini**— автоматическое преобразование `json_schema` → `responseMimeType`/`responseSchema` --**`stream` по умолчанию имеет значение `false`**— соответствует спецификации OpenAI, что позволяет избежать неожиданного SSE в SDK Python/Rust/Go.
+**How OmniRoute solves it:** -<подробности> -🌐 3. «Мой провайдер ИИ блокирует мой регион/страну» +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Такие провайдеры, как OpenAI/Codex, блокируют доступ из определенных географических регионов. Пользователи получают ошибки типа «unsupported_country_region_territory» во время подключений OAuth и API. Особенно это расстраивает разработчиков из развивающихся стран. + -**Как OmniRoute решает эту проблему:** +
+🌐 3. "My AI provider blocks my region/country" --**3-уровневая конфигурация прокси**— настраиваемый прокси-сервер на трех уровнях: глобальный (весь трафик), для каждого провайдера (только один провайдер) и для каждого соединения/ключа. --**Значки прокси с цветной кодировкой**— Визуальные индикаторы: 🟢 глобальный прокси, 🟡 прокси-сервер провайдера, 🔵 прокси-сервер подключения, всегда показывающий IP-адрес. --**Обмен токенами OAuth через прокси**— поток OAuth также проходит через прокси, решая проблему `unsupported_country_region_territory` --**Тесты подключения через прокси**— тесты подключения используют настроенный прокси-сервер (прямого обхода больше нет) --**Поддержка SOCKS5**— Полная поддержка прокси-сервера SOCKS5 для исходящей маршрутизации. --**Подмена отпечатка TLS**— отпечаток TLS, подобный браузеру, через `wreq-js` для обхода обнаружения ботов. --**🔏 Сопоставление отпечатков CLI**— изменяет порядок заголовков и полей тела в соответствии с собственными двоичными сигнатурами CLI, что значительно снижает риск пометки учетной записи. IP-адрес прокси-сервера сохраняется — вы получаете одновременно скрытую**и**маскировку IP-адреса.
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. -<подробности> -🆓 4. «Я хочу использовать ИИ для кодирования, но у меня нет денег» +**How OmniRoute solves it:** -Не каждый может платить 20–200 долларов в месяц за подписку на ИИ. Студентам, разработчикам из развивающихся стран, любителям и фрилансерам нужен доступ к качественным моделям по нулевой цене. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Как OmniRoute решает эту проблему:** + --**Встроенные провайдеры уровня бесплатного пользования**— Встроенная поддержка 100% бесплатных провайдеров: Qoder (5 неограниченных моделей через OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 неограниченных модели: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, Vision-model), Kiro (Claude + AWS Builder ID бесплатно), Gemini CLI (180 тыс. токенов в месяц бесплатно) --**Ollama Cloud**— модели Ollama, размещенные в облаке на api.ollama.com, с бесплатным уровнем «Легкое использование»; используйте префикс `ollamacloud/` --**Комбинации только бесплатно**— Цепочка `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = 0 долларов США в месяц с нулевым временем простоя. --**Бесплатный доступ к NVIDIA NIM**— постоянный бесплатный доступ примерно на 40 об/мин к более чем 70 моделям на build.nvidia.com (переход от кредитов к чистым ограничениям скорости) --**Стратегия оптимизации затрат**— стратегия маршрутизации, которая автоматически выбирает самого дешевого доступного провайдера. +
+🆓 4. "I want to use AI for coding but I have no money" -<подробности> -🔒 5. «Мне нужно защитить свой AI-шлюз от несанкционированного доступа» +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -При предоставлении доступа к сети AI-шлюза (LAN, VPS, Docker) любой, у кого есть адрес, может использовать токены/квоту разработчика. Без защиты API уязвимы для неправильного использования, быстрого внедрения и злоупотреблений. +**How OmniRoute solves it:** -**Как OmniRoute решает эту проблему:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**Управление ключами API**— генерация, ротация и определение области действия для каждого поставщика с помощью специальной страницы `/dashboard/api-manager`. --**Разрешения на уровне модели** — Ограничьте ключи API определенными моделями (`openai/*`, шаблоны подстановочных знаков) с помощью переключателя Разрешить все/Ограничить. --**API Endpoint Protection**— требует ключ для `/v1/models` и блокирует определенных поставщиков из списка. --**Auth Guard + защита CSRF**— все маршруты информационной панели защищены промежуточным программным обеспечением withAuth + токенами CSRF. --**Ограничитель скорости**— ограничение скорости для каждого IP с помощью настраиваемых окон. --**IP-фильтрация**— список разрешенных/блокированных для контроля доступа. --**Prompt Injection Guard**— очистка от вредоносных шаблонов подсказок. --**Шифрование AES-256-GCM**— неактивные учетные данные зашифрованы.
+ -<подробности> -🛑 6. «Мой провайдер вышел из строя, и я потерял процесс кодирования» +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -Поставщики ИИ могут работать нестабильно, возвращать ошибки 5xx или достигать временных ограничений скорости. Если разработчик зависит от одного провайдера, его работу прерывают. Без автоматических выключателей повторные попытки могут привести к сбою приложения. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Как OmniRoute решает эту проблему:** +**How OmniRoute solves it:** --**Выключатель для каждой модели**— автоматическое открытие/закрытие с настраиваемыми пороговыми значениями и временем восстановления (закрыто/открыто/полуоткрыто), область действия зависит от модели, чтобы избежать каскадных блокировок. --**Экспоненциальная задержка** – прогрессивная задержка повторных попыток. --**Anti-Thundering Herd**— Мьютекс + защита семафора от одновременных штормов повторных попыток. --**Комбо-резервные цепочки**— в случае сбоя основного поставщика автоматически проходит через цепочку без вмешательства. --**Комбо-выключатель**— автоматически отключает неисправных поставщиков в комбинированной цепочке. --**Панель состояния**— мониторинг работоспособности, состояния автоматических выключателей, блокировки, статистика кэша, задержка p50/p95/p99.
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest -<подробности> -🔧 7. «Настройка каждого инструмента ИИ утомительна и однообразна» + -Разработчики используют Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Для каждого инструмента требуется своя конфигурация (конечная точка API, ключ, модель). Перенастройка при смене провайдера или модели — пустая трата времени. +
+🛑 6. "My provider went down and I lost my coding flow" -**Как OmniRoute решает эту проблему:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**Панель инструментов CLI**— выделенная страница с настройкой в один клик Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline. --**GitHub Copilot Config Generator**— генерирует `chatLanguageModels.json` для кода VS с массовым выбором модели. --**Мастер адаптации** — пошаговая пошаговая настройка для начинающих пользователей. --**Одна конечная точка, все модели**. Настройте http://localhost:20128/v1 один раз и получите доступ к более чем 60 провайдерам.
+**How OmniRoute solves it:** -<подробности> -🔑 8. «Управление токенами OAuth от нескольких провайдеров — это ад» +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot — все используют OAuth 2.0 с токенами с истекающим сроком действия. Разработчикам необходимо постоянно проходить повторную аутентификацию, решать проблемы «client_secret отсутствует», «redirect_uri_mismatch» и сбои на удаленных серверах. OAuth в LAN/VPS особенно проблематичен. + -**Как OmniRoute решает эту проблему:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Автоматическое обновление токенов**— токены OAuth обновляются в фоновом режиме до истечения срока их действия. --**Встроенный OAuth 2.0 (PKCE)**— автоматический поток для Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder. --**OAuth с несколькими учетными записями** – несколько учетных записей для каждого провайдера посредством извлечения токена JWT/ID. --**OAuth LAN/Remote Fix**— обнаружение частного IP-адреса для redirect_uri + ручной режим URL-адреса для удаленных серверов. --**OAuth за Nginx**— использует window.location.origin для совместимости с обратным прокси-сервером. --**Руководство по удаленному OAuth**— пошаговое руководство по учетным данным Google Cloud на VPS/Docker.
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. -<подробности> -📊 9. «Я не знаю, сколько и куда я трачу» +**How OmniRoute solves it:** -Разработчики используют нескольких платных поставщиков, но не имеют единого представления о расходах. У каждого провайдера есть своя панель выставления счетов, но единого представления нет. Неожиданные расходы могут накопиться. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Как OmniRoute решает эту проблему:** + --**Панель анализа затрат**— отслеживание затрат на каждый токен и управление бюджетом для каждого поставщика. --**Ограничения бюджета на уровень** — потолок расходов на уровень, который запускает автоматический возврат к резервному варианту. --**Конфигурация цен на модель**— настраиваемые цены на модель. --**Статистика использования каждого ключа API**— количество запросов и временная метка последнего использования для каждого ключа. --**Панель аналитики**— карточки статистики, диаграмма использования модели, таблица поставщиков с показателями успеха и задержкой. +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" -<подробности> -🐛 10. «Я не могу диагностировать ошибки и проблемы в вызовах ИИ» +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Когда вызов завершается неудачей, разработчик не знает, было ли это ограничение скорости, срок действия токена, неправильный формат или ошибка провайдера. Фрагментированные журналы на разных терминалах. Без наблюдаемости отладка осуществляется методом проб и ошибок. +**How OmniRoute solves it:** -**Как OmniRoute решает эту проблему:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Панель управления унифицированными журналами**— 4 вкладки: журналы запросов, журналы прокси, журналы аудита, консоль. --**Консольный просмотр журнала**— просмотрщик в режиме терминала в режиме реального времени с уровнями с цветовой кодировкой, автоматической прокруткой, поиском и фильтрацией. --**Журналы прокси-сервера SQLite**— постоянные журналы, которые сохраняются после перезапуска сервера. --**Площадка переводчика**— 4 режима отладки: Площадка (перевод формата), Тестер чата (туда и обратно), Тестовый стенд (пакетный), Мониторинг в реальном времени (в режиме реального времени). --**Запрос телеметрии**— задержка p50/p95/p99 + отслеживание X-Request-Id --**Файловое ведение журнала с ротацией**— журналы приложений чередуются по размеру, дням хранения и количеству архивов; Артефакты журнала вызовов чередуются по дням хранения и количеству файлов --**Отчет о информации о системе**— `npm run system-info` генерирует `system-info.txt` с вашей полной средой (версия узла, версия OmniRoute, ОС, инструменты CLI, статус Docker/PM2). Прикрепите его при сообщении о проблемах для мгновенной сортировки.
+ -<подробности> -🏗️ 11. «Развертывание и обслуживание шлюза сложны» +
+📊 9. "I don't know how much I'm spending or where" -Установка, настройка и обслуживание прокси-сервера AI в различных средах (локальных, VPS, Docker, облаке) — трудоемкий процесс. Такие проблемы, как жестко запрограммированные пути, EACCES в каталогах, конфликты портов и кроссплатформенные сборки, добавляют проблем. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Как OmniRoute решает эту проблему:** +**How OmniRoute solves it:** --**npm global install**— `npm install -g omniroute && omniroute` — готово --**Мультиплатформенность Docker**— встроенная версия AMD64 + ARM64 (Apple Silicon, AWS Graviton, Raspberry Pi) --**Профили Docker Compose**— `base` (без инструментов CLI) и `cli` (с Claude Code, Codex, OpenClaw). --**Electron Desktop App**— собственное приложение для Windows/macOS/Linux с панелью задач, автозапуском и автономным режимом. --**Режим разделения портов**— API и панель мониторинга на отдельных портах для расширенных сценариев (обратный прокси-сервер, сеть контейнеров). --**Cloud Sync**— синхронизация конфигурации между устройствами через Cloudflare Workers. --**Резервные копии БД**— автоматическое резервное копирование, восстановление, экспорт и импорт всех настроек с помощью DISABLE_SQLITE_AUTO_BACKUP для резервных копий, управляемых извне.
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency -<подробности> -🌍 12. «Интерфейс только на английском языке, а моя команда не говорит по-английски» + -Команды в неанглоязычных странах, особенно в Латинской Америке, Азии и Европе, испытывают трудности с интерфейсами только на английском языке. Языковые барьеры сокращают внедрение и увеличивают количество ошибок в конфигурации. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Как OmniRoute решает эту проблему:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Панель управления i18n — 30 языков**— Все более 500 клавиш переведены, включая арабский, болгарский, датский, немецкий, испанский, финский, французский, иврит, хинди, венгерский, индонезийский, итальянский, японский, корейский, малайский, голландский, норвежский, польский, португальский (PT/BR), румынский, русский, словацкий, шведский, тайский, украинский, вьетнамский, китайский, филиппинский, английский --**Поддержка RTL**— поддержка написания справа налево для арабского языка и иврита. --**Многоязычные файлы README**— 30 полных переводов документации. --**Выбор языка**— значок глобуса в заголовке для переключения в реальном времени.
+**How OmniRoute solves it:** -<подробности> -🔄 13. «Мне нужно больше, чем просто чат — мне нужны встраивания, изображения, аудио» +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -ИИ — это не просто завершение чата. Разработчикам необходимо генерировать изображения, расшифровывать аудио, создавать вложения для RAG, изменять ранжирование документов и модерировать контент. Каждый API имеет свою конечную точку и формат. + -**Как OmniRoute решает эту проблему:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Embeddings**— `/v1/embeddings` с 6 поставщиками и более чем 9 моделями. --**Генерация изображений**— `/v1/images/generations` с 10 поставщиками и более чем 20 моделями (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Преобразование текста в видео** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) и SD WebUI. --**Преобразование текста в музыку**— `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) --**Транскрипция аудио**— `/v1/audio/transcriptions` – Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Преобразование текста в речь**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + существующие поставщики --**Модерации**— `/v1/moderations` — Проверки безопасности контента. --**Реранжирование**— `/v1/rerank` — изменение ранжирования релевантности документа. --**Responses API**— Полная поддержка `/v1/responses` для Кодекса.
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. -<подробности> -🧪 14. «У меня нет возможности тестировать и сравнивать качество разных моделей» +**How OmniRoute solves it:** -Разработчики хотят знать, какая модель лучше всего подходит для их варианта использования (код, перевод, рассуждения), но сравнение вручную занимает много времени. Интегрированных инструментов оценки не существует. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**Как OmniRoute решает эту проблему:** + --**Оценки LLM** — тестирование золотого набора с 10 предварительно загруженными вариантами, охватывающими приветствия, математику, географию, генерацию кода, соответствие JSON, перевод, уценку, отказ от безопасности. --**4 стратегии сопоставления**— `точные`, `содержит`, `регулярное выражение`, `пользовательское` (функция JS). --**Тестовый стенд Translator Playground**— пакетное тестирование с несколькими входными данными и ожидаемыми результатами, сравнение между поставщиками. --**Тестер чата** — полный цикл с визуальным отображением ответов. --**Живой монитор**— поток всех запросов, проходящих через прокси, в реальном времени. +
+🌍 12. "The interface is English-only and my team doesn't speak English" -<подробности> -📈 15. «Мне нужно масштабироваться без потери производительности» +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -По мере роста объема запросов без кэширования одних и тех же вопросов возникают дублирующие затраты. Без идемпотентности дублирование запросов приводит к отходам обработки. Необходимо соблюдать ограничения по тарифам для каждого поставщика. +**How OmniRoute solves it:** -**Как OmniRoute решает эту проблему:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Семантический кеш**— двухуровневый кеш (сигнатура + семантика) снижает стоимость и задержку. --**Request Idempotency**— окно дедупликации 5 с для идентичных запросов. --**Обнаружение ограничения скорости**— число оборотов в минуту для каждого провайдера, минимальный разрыв и максимальное одновременное отслеживание. --**Редактируемые ограничения скорости**— настраиваемые значения по умолчанию в меню «Настройки» → «Устойчивость с постоянством». --**Кэш проверки ключей API**— трехуровневый кеш для повышения производительности. --**Панель состояния с телеметрией**— задержка p50/p95/p99, статистика кэша, время безотказной работы.
+ -<подробности> -🤖 16. «Я хочу глобально контролировать поведение модели» +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Разработчики, которым нужны все ответы на определенном языке, с определенным тоном или которые хотят ограничить количество токенов рассуждения. Настраивать это в каждом инструменте/запросе непрактично. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Как OmniRoute решает эту проблему:** +**How OmniRoute solves it:** --**Внедрение системных подсказок**— глобальное приглашение применяется ко всем запросам. --**Продуманная проверка бюджета**— контроль распределения токенов для каждого запроса (сквозной, автоматический, пользовательский, адаптивный) --**9 стратегий маршрутизации**— глобальные стратегии, определяющие распределение запросов. --**Wildcard Router**— шаблоны `provider/*` динамически маршрутизируются к любому провайдеру. --**Переключение/включение комбо**— переключение комбо непосредственно с панели управления. --**Переключение поставщика**— включение/отключение всех подключений к провайдеру одним щелчком мыши. --**Заблокированные поставщики**— исключить определенных поставщиков из списка `/v1/models`.
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex -<подробности> -🧰 17. «Мне нужны инструменты MCP как первоклассные возможности продукта» + -Многие шлюзы AI предоставляют MCP только как скрытую деталь реализации. Командам нужен видимый и управляемый операционный уровень. +
+🧪 14. "I have no way to test and compare quality across models" -**Как OmniRoute решает эту проблему:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP отображается на панели навигации панели управления и на вкладке протокола конечной точки. -- Отдельная страница управления MCP с процессами, инструментами, объемами работ и аудитом. -- Встроенный быстрый старт для `omniroute --mcp` и подключения клиента.
+**How OmniRoute solves it:** -<подробности> -🧠 18. «Мне нужна оркестрация A2A с путями задач синхронизации и потоковой передачи» +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Рабочие процессы агента требуют как прямых ответов, так и длительного потокового выполнения с контролем жизненного цикла. + -**Как OmniRoute решает эту проблему:** +
+📈 15. "I need to scale without losing performance" -- Конечная точка A2A JSON-RPC (`POST/a2a`) с `message/send` и `message/stream` -- Потоковая передача SSE с распространением состояния терминала -- API жизненного цикла задач для задач/получить и задач/отмены.
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. -<подробности> -🛰️ 19. «Мне нужно реальное состояние процесса MCP, а не угаданный статус» +**How OmniRoute solves it:** -Оперативным группам необходимо знать, действительно ли MCP работает, а не только доступен ли API. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**Как OmniRoute решает эту проблему:** + -- Файл контрольного сигнала времени выполнения с PID, метками времени, транспортом, количеством инструментов и режимом области действия. -- API статуса MCP, объединяющий пульс + недавнюю активность -- Карты состояния пользовательского интерфейса для актуальности процессов, времени безотказной работы и пульса. +
+🤖 16. "I want to control model behavior globally" -<подробности> -📋 20. «Мне нужно проверяемое выполнение инструмента MCP» +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Когда инструменты изменяют конфигурацию или запускают действия операционной системы, командам необходима судебно-медицинская отслеживаемость. +**How OmniRoute solves it:** -**Как OmniRoute решает эту проблему:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- Ведение журнала аудита на основе SQLite для вызовов инструментов MCP. -- Фильтры по инструменту, успеху/неуспеху, ключу API и нумерации страниц. -- Таблица аудита панели мониторинга + конечные точки статистики для автоматизации
+ -<подробности> -🔐 21. «Мне нужны ограниченные разрешения MCP для каждой интеграции» +
+🧰 17. "I need MCP tools as first-class product capabilities" -Разные клиенты должны иметь минимальный доступ к категориям инструментов. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Как OmniRoute решает эту проблему:** +**How OmniRoute solves it:** -- 10 детальных областей MCP для контролируемого доступа к инструментам -- Обеспечение соблюдения границ и видимость в пользовательском интерфейсе управления MCP. -- Безопасное положение по умолчанию для рабочих инструментов.
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding -<подробности> -⚙️ 22. «Мне нужен оперативный контроль без передислокации» + -Командам необходимы быстрые изменения во время выполнения во время инцидентов или событий, связанных с затратами. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**Как OmniRoute решает эту проблему:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Переключение комбо-активации прямо с панели управления MCP. -- Применение профилей устойчивости из предварительно определенных пакетов политик. -- Сброс состояния автоматического выключателя с той же панели управления.
+**How OmniRoute solves it:** -<подробности> -🔄 23. «Мне нужна оперативная видимость жизненного цикла задачи A2A и ее отмена» +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Без прозрачности жизненного цикла инциденты с задачами становится трудно сортировать. + -**Как OmniRoute решает эту проблему:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Список задач/фильтрация по состоянию/навыку с нумерацией страниц -- Детализация метаданных задачи, событий и артефактов. -- Конечная точка отмены задачи и действие пользовательского интерфейса с подтверждением.
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. -<подробности> -🌊 24. «Мне нужны метрики активного потока для загрузки A2A» +**How OmniRoute solves it:** -Рабочие процессы потоковой передачи требуют оперативного понимания параллелизма и живых соединений. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**Как OmniRoute решает эту проблему:** + -- Счетчики активных потоков интегрированы в статус A2A -- Временная метка последней задачи и количество состояний -- Карты информационной панели A2A для мониторинга операций в реальном времени. +
+📋 20. "I need auditable MCP tool execution" -<подробности> -🪪 25. «Мне нужно стандартное обнаружение агентов для клиентов» +When tools mutate config or trigger ops actions, teams need forensic traceability. -Внешним клиентам и оркестраторам для адаптации необходимы машиночитаемые метаданные. +**How OmniRoute solves it:** -**Как OmniRoute решает эту проблему:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- Карта агента доступна в `/.well-known/agent.json` -- Возможности и навыки, отображаемые в пользовательском интерфейсе управления. -- API статуса A2A включает метаданные обнаружения для автоматизации.
+ -<подробности> -🧭 26. «Мне нужна возможность обнаружения протоколов в UX продукта» +
+🔐 21. "I need scoped MCP permissions per integration" -Если пользователи не могут обнаружить поверхности протокола, качество внедрения и поддержки снижается. +Different clients should have least-privilege access to tool categories. -**Как OmniRoute решает эту проблему:** +**How OmniRoute solves it:** -- Объединенная страница**Конечные точки**с вкладками для конечных точек прокси, MCP, A2A и API. -- Переключение статуса встроенного сервиса (Онлайн/Офлайн) для MCP и A2A. -- Ссылки из обзора на специальные вкладки управления.
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling -<подробности> -🧪 27. «Мне нужна сквозная проверка протокола с реальными клиентами» + -Пробных тестов недостаточно для проверки совместимости протокола перед выпуском. +
+⚙️ 22. "I need operational controls without redeploying" -**Как OmniRoute решает эту проблему:** +Teams need quick runtime changes during incidents or cost events. -- Пакет E2E, который загружает приложение и использует настоящий клиентский транспорт MCP SDK. -- Клиент A2A тестирует потоки обнаружения, отправки, потоковой передачи, получения и отмены. -- Перекрестная проверка утверждений с помощью API-интерфейсов аудита MCP и задач A2A.
+**How OmniRoute solves it:** -<подробности> -📡 28. «Мне нужна унифицированная наблюдаемость на всех интерфейсах» +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -Разделение наблюдаемости по протоколам создает «слепые зоны» и увеличивает MTTR. + -**Как OmniRoute решает эту проблему:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Унифицированные дашборды/логи/аналитика в одном продукте -- Здоровье + аудит + телеметрия запросов на уровнях OpenAI, MCP и A2A. -- Операционные API для статуса и автоматизации
+Without lifecycle visibility, task incidents become hard to triage. -<подробности> -💼 29. «Мне нужна одна среда выполнения для прокси + инструментов + оркестровки агентов» +**How OmniRoute solves it:** -Запуск множества отдельных служб увеличивает эксплуатационные расходы и количество видов сбоев. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**Как OmniRoute решает эту проблему:** + -- OpenAI-совместимый прокси, сервер MCP и сервер A2A в одном стеке -- Общая аутентификация, устойчивость, хранилище данных и наблюдаемость. -- Согласованная модель политики на всех поверхностях взаимодействия. +
+🌊 24. "I need active stream metrics for A2A load" -<подробности> -🚀 30. «Мне нужно реализовать агентские рабочие процессы без разрастания связующего кода» +Streaming workflows require operational insight into concurrency and live connections. -Команды теряют скорость при объединении нескольких специальных сервисов и сценариев. +**How OmniRoute solves it:** -**Как OmniRoute решает эту проблему:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Единая стратегия конечных точек для клиентов и агентов -- Встроенные пользовательские интерфейсы управления протоколами и пути проверки дыма. -- Готовые к работе основы (безопасность, ведение журналов, отказоустойчивость, резервное копирование)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Пособие А: максимальное использование платной подписки + дешевое резервное копирование**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -608,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Пособие Б: стек кодирования с нулевой стоимостью**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Пособие C: Всегда работающая резервная цепочка 24 часа в сутки, 7 дней в неделю**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -631,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Сборник D: Операции агента с помощью MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Настройте ИИ-кодирование за считанные минуты по цене**0 долларов США в месяц**. Подключите эти бесплатные учетные записи и используйте встроенную комбинацию**Free Stack**. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Шаг | Действие | Провайдеры разблокированы | +| Step | Action | Providers Unlocked | | ---- | -------------------------------------------------- | ------------------------------------------------------------------ | -| 1 | Подключите**Kiro**(идентификатор AWS Builder ID OAuth) | Клод Сонет 4.5, Haiku 4.5 —**без ограничений**| -| 2 | Подключите**Qoder**(Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... —**без ограничений**| -| 3 | Подключите**Qwen**(Код устройства) | qwen3-coder-plus, qwen3-coder-flash... —**без ограничений**| -| 4 | Подключите**Gemini CLI**(Google OAuth) | Gemini-3-flash, Gemini-2.5-pro —**180 тысяч в месяц бесплатно**| -| 5 | `/dashboard/combos` →**Бесплатный шаблон стека ($0)**| Автоматический циклический перебор всех бесплатных провайдеров | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Укажите в любой IDE/CLI:**`http://localhost:20128/v1` · Ключ API: `any-string` · Готово. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Дополнительное покрытие (также бесплатно):**Ключ API Groq (30 об/мин бесплатно), NVIDIA NIM (40 об/мин бесплатно, более 70 моделей), Cerebras (1 млн токов в день), ключ API LongCat (50 млн токенов в день!), Cloudflare Workers AI (10 тыс. нейронов в день, более 50 моделей).## Быстрый старт +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Быстрый старт ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **Пользователи pnpm:**Запустите `pnpm Approved-builds -g` после установки, чтобы включить собственные сценарии сборки, необходимые `better-sqlite3` и `@swc/core`: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > -> ```баш +> ```bash > pnpm install -g omniroute -> pnpm Approval-Builds -g # Выбрать все пакеты → утвердить -> всенаправленный маршрут +> pnpm approve-builds -g # Select all packages → approve +> omniroute > ``` -Панель мониторинга открывается по адресу http://localhost:20128, а базовый URL-адрес API — http://localhost:20128/v1. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Команда | Описание | -| ----------------------- | ----------------------------------------------------------------------- | -| `всенаправление` | Запустить сервер (`PORT=20128`, API и панель управления на одном порту) | -| `omniroute --port 3000` | Установите канонический порт/порт API на 3000 | -| `omniroute --mcp` | Запустить сервер MCP (транспорт stdio) | -| `omniroute --no-open` | Не открывать браузер автоматически | -| `omniroute --help` | Показать помощь | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Дополнительный режим разделения портов:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -Для большинства развертываний вам нужно только: +For most deployments, you only need: -| Переменная | По умолчанию | Цель | +| Variable | Default | Purpose | | ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600000` | Общая базовая линия для восходящей выборки, скрытых тайм-аутов Undici, запросов отпечатков пальцев TLS и тайм-аутов запросов моста API/прокси | -| `STREAM_IDLE_TIMEOUT_MS` | наследует `REQUEST_TIMEOUT_MS` | Максимальный промежуток между фрагментами потоковой передачи, прежде чем OmniRoute прервет поток SSE | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -Обратная совместимость сохраняется: существующие `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` и другие переменные таймаута для каждого уровня по-прежнему работают и переопределяют общий базовый уровень. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Если вам нужен более точный контроль, доступны расширенные переопределения:| Переменная | По умолчанию | Цель | -| ---------------------------------------- | ----------------------------------------- | -------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | наследует `REQUEST_TIMEOUT_MS` | Общий тайм-аут восходящего запроса, используемый основным сигналом прерывания выборки | -| `FETCH_HEADERS_TIMEOUT_MS` | наследует `FETCH_TIMEOUT_MS` | Ограничение времени Undici для получения заголовков ответов восходящего потока | -| `FETCH_BODY_TIMEOUT_MS` | наследует `FETCH_TIMEOUT_MS` | Ограничение времени Undici между восходящими фрагментами тела (`0` отключает его) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Тайм-аут TCP-соединения Undici | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Тайм-аут сокета бездействия Undici | -| `TLS_CLIENT_TIMEOUT_MS` | наследует `FETCH_TIMEOUT_MS` | Таймаут для запросов отпечатков пальцев TLS, сделанных через `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | наследует `REQUEST_TIMEOUT_MS` или `30000` | Таймаут для переадресации прокси `/v1` с порта API на порт информационной панели | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `макс(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Тайм-аут входящего запроса на сервере моста API | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Тайм-аут входящего заголовка на сервере моста API | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Таймаут поддержания активности на сервере моста API | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Тайм-аут бездействия сокета на сервере моста API (`0` отключает его) | +Advanced overrides are available if you need finer control: -Если вы запускаете OmniRoute за Nginx, Caddy, Cloudflare или другим обратным прокси-сервером, убедитесь, что прокси-сервер -таймауты также превышают таймауты потока/выборки OmniRoute.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Откройте Dashboard → «Провайдеры» и подключите хотя бы одного провайдера (OAuth или ключ API). -2. Откройте Dashboard → «Конечные точки» и создайте ключ API. -3. (Необязательно) Откройте Панель управления → «Комбо» и установите резервную цепочку.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Работает с Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode и OpenAI-совместимыми SDK.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (для операций с использованием инструмента):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -Затем подключите клиент MCP через stdio и протестируйте такие инструменты, как: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (для рабочих процессов между агентами):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -760,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Этот пакет проверяет реальные клиентские потоки MCP и A2A на соответствие работающему приложению.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -768,14 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` -<подробности> +
+Void Linux (`xbps-src` template) -Void Linux (шаблон `xbps-src`) - -Пользователи Void Linux могут создать собственный пакет с помощью xbps-src. Сохраните этот блок как srcpkgs/omniroute/template:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -787,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -795,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -871,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -882,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute доступен как общедоступный образ Docker на [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Быстрый запуск:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -892,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**С файлом среды:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Использование Docker Compose:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Поддержка информационной панели для развертываний Docker теперь включает в себя**Cloudflare Quick Tunnel**в один клик на панели инструментов → Конечные точки. Первое разрешение загружает `cloudflare` только при необходимости, запускает временный туннель к вашей текущей конечной точке `/v1` и показывает сгенерированный URL-адрес `https://*.trycloudflare.com/v1` непосредственно под вашим обычным общедоступным URL-адресом. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Примечания: +Notes: -- URL-адреса быстрого туннеля являются временными и изменяются после каждого перезапуска. -- Быстрые туннели не восстанавливаются автоматически после перезапуска OmniRoute или контейнера. Re-enable them from the dashboard when needed. -- Управляемая установка в настоящее время поддерживает Linux, macOS и Windows на x64/arm64. - — Управляемые быстрые туннели по умолчанию используют транспорт HTTP/2, чтобы избежать шумных предупреждений буфера QUIC UDP в ограниченных контейнерных средах. Установите `CLOUDFLARED_PROTOCOL=quic` или `auto`, если вам нужен другой транспорт. -- Образы Docker объединяют корни системного центра сертификации и передают их в управляемый «cloudflared», что позволяет избежать сбоев доверия TLS при загрузке туннеля внутри контейнера. -- SQLite работает в режиме WAL. `docker stop` должен завершиться, чтобы OmniRoute мог проверить последние изменения обратно в `storage.sqlite`. -- В комплекте файлов Compose уже установлен льготный период остановки в 40 секунд. Если вы запускаете образ напрямую, оставьте `--stop-timeout 40` (или подобное), чтобы ручная остановка не прерывала очистку при завершении работы. -- Установите `CLOUDFLARED_BIN=/absolute/path/to/cloudflared`, если вы хотите, чтобы OmniRoute использовал существующий двоичный файл вместо его загрузки. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Использование Docker Compose с Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute можно безопасно открыть с помощью автоматического обеспечения SSL Caddy. Убедитесь, что запись DNS A вашего домена указывает на IP-адрес вашего сервера.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` - -| Изображение | Тег | Размер | Описание | +| Image | Tag | Size | Description | | ------------------------ | -------- | ------ | --------------------- | -| `diegosouzapw/omniroute` | `последний` | ~250 МБ | Последняя стабильная версия | -| `diegosouzapw/omniroute` | `1.0.3` | ~250 МБ | Текущая версия |--- +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | + +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**НОВИНКА!**OmniRoute теперь доступен как**собственное настольное приложение**для Windows, macOS и Linux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Запускайте OmniRoute как отдельное настольное приложение — для локальных моделей не требуется ни терминала, ни браузера, ни Интернета. Приложение на базе Electron включает в себя: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Встроенное окно**— выделенное окно приложения с интеграцией в системный трей. -- 🔄**Автозапуск**— запуск OmniRoute при входе в систему. -- 🔔**Встроенные уведомления**— получайте оповещения об исчерпании квоты или проблемах с поставщиком услуг. -- ⚡**Установка в один клик**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Офлайн-режим**— работает полностью в автономном режиме со встроенным сервером.### Быстрый старт +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Быстрый старт ```bash # Development mode @@ -981,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -В свернутом виде OmniRoute находится на панели задач и предлагает быстрые действия: +When minimized, OmniRoute lives in your system tray with quick actions: -- Открыть панель управления -- Изменить порт сервера -- Закрыть приложение +- Open dashboard +- Change server port +- Quit application -📖 Полная документация: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Уровень | Провайдер | Cost | Сброс квоты | Лучшее для | -| ---------------- | --------------------------- | -------------------------------------------------- | ---------------------------- | ---------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 ПОДПИСКА** | Клод Код (Про) | 20 долларов США в месяц | 5 часов + еженедельно | Уже подписан | -| | Кодекс (Плюс/Про) | 20–200 долларов в месяц | 5 часов + еженедельно | Пользователи OpenAI | -| | Близнецы CLI | **БЕСПЛАТНО** | 180 тыс./мес + 1 тыс./день | Каждый! | -| | Второй пилот GitHub | 10–19 долларов в месяц | Ежемесячно | Пользователи GitHub | -| **🔑 КЛЮЧ API** | NVIDIA НИМ | **БЕСПЛАТНО**(для разработчиков навсегда) | ~40 об/мин | 70+ открытых моделей | -| | Церебра | **БЕСПЛАТНО**(1 миллион ток/день) | 60 000 т/мин / 30 об/мин | Самый быстрый в мире | -| | Грок | **БЕСПЛАТНО**(30 об/мин) | 14,4К РПД | Сверхбыстрая Лама/Джемма | -| | ДипСик V3.2 | 0,27 доллара США/1,10 доллара США за 1 миллион | Нет | Лучшее соотношение цены и качества | -| | xAI Грок-4 Быстрый | **0,20 доллара США/0,50 доллара США за 1 месяц**🆕 | Нет | Самый быстрый + вызов инструмента, сверхнизкий | -| | xAI Грок-4 (стандартный) | 0,20 доллара США/1,50 доллара США за 1 миллион 🆕 | Нет | Рассуждающий флагман от xAI | -| | Мистраль | Бесплатная пробная версия + платная | Скорость ограничена | Европейский ИИ | -| | OpenRouter | Оплата по факту использования | Нет | 100+ моделей общ. | -| **💰 ДЕШЕВО** | GLM-5 (через Z.AI) 🆕 | 0,5 долл. США/1 млн | Ежедневно в 10:00 | Выход 128K, новейший флагман | -| | ГЛМ-4.7 | 0,6 долл. США/1 млн | Ежедневно в 10:00 | Резервное копирование бюджета | -| | МиниМакс М2.5 🆕 | Вклад $0,3/1 млн | 5-часовой прокат | Рассуждение + агентные задачи | -| | МиниМакс М2.1 | 0,2 долл. США/1 млн | 5-часовой прокат | Самый дешевый вариант | -| | Кими K2.5 (Moonshot API) 🆕 | Оплата по факту использования | Нет | Прямой доступ к Moonshot API | -| | Кими К2 | $9/mo flat | 10 миллионов токенов в месяц | Предсказуемая стоимость | -| **🆓 БЕСПЛАТНО** | Кодер | **$0** | Неограниченный | 5 моделей без ограничений | -| | Квен | **$0** | Неограниченный | 4 модели без ограничений | -| | Киро | **$0** | Неограниченный | Клод Сонет/Haiku (AWS Builder) | -| | LongCat Flash-Lite 🆕 | **0$**(50 миллионов токов в день 🔥) | 1 ЗП | Самая большая бесплатная квота на Земле | -| | Опыление AI 🆕 | **$0**(ключ не требуется) | 1 запрос/15 с | GPT-5, Клод, DeepSeek, Лама 4 | -| | Рабочие Cloudflare AI 🆕 | **$0**(10 тыс. нейронов в день) | ~150 представителей/день | Более 50 моделей, глобальное преимущество | -| | Scaleway AI 🆕 | **$0**(всего 1 млн токенов) | Скорость ограничена | ЕС/GDPR, Qwen3 235B, Лама 70B | > 🆕**Добавлены новые модели (март 2026 г.):**Семейство Grok-4 Fast по цене 0,20 долл. США/0,50 долл. США/млн (тестовое время 1143 мс — на 30 % быстрее, чем Gemini 2.5 Flash), GLM-5 через Z.AI с выходом 128 КБ, рассуждения MiniMax M2.5, обновленная цена DeepSeek V3.2, Kimi K2.5 через Moonshot Direct API. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 Комбо-стек за 0 долларов США — полная бесплатная установка:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Нулевая стоимость. Кодирование никогда не прекращается.**Настройте это как одну комбинацию OmniRoute, и все резервные варианты будут выполняться автоматически — без ручного переключения.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Все представленные ниже модели**на 100 % бесплатны, при этом кредитная карта не требуется**. OmniRoute автоматически выполняет маршрутизацию между ними, когда заканчивается одна квота — объедините их все для получения нерушимой комбинации стоимостью 0 долларов.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Модель | Префикс | Лимит | Ограничение ставки | +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | --------------------- | -| `Клод-сонет-4.5` | `кр/` |**Неограниченно**| Нет сообщений о дневном лимите | -| `Клод-Хайку-4.5` | `кр/` |**Неограниченно**| Нет сообщений о дневном лимите | -| `Клод-опус-4.6` | `кр/` |**Неограниченно**| Последний опус через Киро |### 🟢 QODER MODELS (Free PAT via qodercli) +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -| Модель | Префикс | Лимит | Ограничение ставки | +### 🟢 QODER MODELS (Free PAT via qodercli) + +| Model | Prefix | Limit | Rate Limit | | ------------------ | ------ | ------------- | --------------- | -| `кими-k2-мышление` | `если/` |**Неограниченно**| Ограничение не сообщается | -| `qwen3-кодер-плюс` | `если/` |**Неограниченно**| Ограничение не сообщается | -| `глубокий поиск-r1` | `если/` |**Неограниченно**| Ограничение не сообщается | -| `минимакс-м2.1` | `если/` |**Неограниченно**| Ограничение не сообщается | -| `кими-к2` | `если/` |**Неограниченно**| Ограничение не сообщается | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -> Рекомендуемый метод подключения:**Токен личного доступа + `qodercli`**. OAuth браузера — это -> экспериментальный и отключен по умолчанию, если не настроены переменные среды `QODER_OAUTH_*`.### 🟡 QWEN MODELS (Device Code Auth) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| Модель | Префикс | Лимит | Ограничение ставки | +### 🟡 QWEN MODELS (Device Code Auth) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | ------------------- | -| `qwen3-кодер-плюс` | `qw/` |**Неограниченно**| Ограничение не сообщается | -| `qwen3-coder-flash` | `qw/` |**Неограниченно**| Ограничение не сообщается | -| `qwen3-coder-next` | `qw/` |**Неограниченно**| Ограничение не сообщается | -| `видение-модель` | `qw/` |**Неограниченно**| Мультимодальный (изображения) |### 🟣 GEMINI CLI (Google OAuth) +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Модель | Префикс | Лимит | Ограничение ставки | +### 🟣 GEMINI CLI (Google OAuth) + +| Model | Prefix | Limit | Rate Limit | | ------------------------ | ------ | --------------------------- | ------------- | -| `gemini-3-flash-preview` | `gc/` |**180 тыс. ток/месяц**+ 1 тыс. ток/день | Ежемесячный сброс | -| `Близнецы-2.5-про` | `gc/` | 180 тыс. в месяц (общий пул) | Высокое качество |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -| Уровень | Дневной лимит | Ограничение ставки | Заметки | +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---------- | ------------ | ----------- | ------------------------------------------------------ | -| Бесплатно (Для разработчиков) | Нет ограничения токена |**~40 об/мин**| 70+ моделей; переход на чистые лимиты ставок в середине 2025 года | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -Популярные бесплатные модели: «moonshotai/kimi-k2.5» (Kimi K2.5), «z-ai/glm4.7» (GLM 4.7), «deepseek-ai/deepseek-v3.2» (DeepSeek V3.2), «nvidia/llama-3.3-70b-instruct», «deepseek/deepseek-r1»### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -| Уровень | Дневной лимит | Ограничение ставки | Заметки | +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) + +| Tier | Daily Limit | Rate Limit | Notes | | ---- | ----------------- | ---------------- | ------------------------------------------- | -| Бесплатно |**1 млн токенов в день**| 60 000 т/мин / 30 об/мин | Самый быстрый в мире вывод LLM; сбрасывается ежедневно | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -Доступно бесплатно: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -| Уровень | Дневной лимит | Ограничение ставки | Заметки | +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---- | ------------- | ---------------- | ----------------------------------------- | -| Бесплатно |**14,4 тыс. РПД**| 30 об/мин на модель | Нет кредитной карты; 429 по лимиту, не взимается | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -Доступны бесплатно: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -| Модель | Префикс | Ежедневная бесплатная квота | Заметки | +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 + +| Model | Prefix | Daily Free Quota | Notes | | ----------------------------- | ------ | ----------------- | ----------------------- | -| `LongCat-Flash-Lite` | `ЖК/` |**50 миллионов токенов**💥 | Самая большая бесплатная квота за всю историю | -| `LongCat-Flash-Chat` | `ЖК/` | 500 тыс. жетонов | Многоходовой чат | -| `LongCat-Flash-Thinking` | `ЖК/` | 500 тыс. жетонов | Рассуждение / ЦТ | -| `LongCat-Flash-Thinking-2601` | `ЖК/` | 500 тыс. жетонов | Версия от января 2026 г. | -| `LongCat-Flash-Omni-2603` | `ЖК/` | 500 тыс. жетонов | Мультимодальный | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | -> 100% бесплатно во время публичной бета-версии. Зарегистрируйтесь на [longcat.chat](https://longcat.chat) по электронной почте или телефону. Сбрасывается ежедневно в 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. -| Модель | Префикс | Ограничение ставки | Поставщик позади | +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| `опенай` | `пол/` | 1 запрос/15 с | ГПТ-5 | -| `Клод` | `пол/` | 1 запрос/15 с | Антропный Клод | -| `близнецы` | `пол/` | 1 запрос/15 с | Google Близнецы | -| `глубокий поиск` | `пол/` | 1 запрос/15 с | ДипСик V3 | -| `лама` | `пол/` | 1 запрос/15 с | Мета Лама 4 Разведчик | -| `мистраль` | `пол/` | 1 запрос/15 с | Мистраль ИИ | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**Нет проблем:**нет регистрации и нет ключа API. Добавьте поставщика Pollinations с пустым ключевым полем, и он сразу заработает.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| Уровень | Ежедневные нейроны | Эквивалентное использование | Заметки | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 + +| Tier | Daily Neurons | Equivalent Usage | Notes | | ---- | ------------- | --------------------------------------- | ----------------------- | -| Бесплатно |**10 000**| ~ 150 LLM соответственно / 500 с аудио / 15 тыс. вставок | Глобальное преимущество, более 50 моделей | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -Популярные бесплатные модели: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (бесплатное аудио!), `@cf/qwen/qwen2.5-coder-15b-instruct` +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -> Требуется токен API + идентификатор учетной записи от [dash.cloudflare.com](https://dash.cloudflare.com). Сохраните идентификатор учетной записи в настройках провайдера.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -| Уровень | Бесплатная квота | Местоположение | Заметки | -| ---- | ------------- | ------------ | -------------------- | -| Бесплатно |**1 млн токенов**| 🇫🇷 Париж, ЕС | Кредитная карта не требуется в определенных пределах | +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 -Доступно бесплатно: qwen3-235b-a22b-instruct-2507 (Qwen3 235B!), llama-3.1-70b-instruct, mistral-small-3.2-24b-instruct-2506, deepseek-v3-0324. +| Tier | Free Quota | Location | Notes | +| ---- | ------------- | ------------ | ----------------------------------- | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | -> Соответствует ЕС/GDPR. Получите ключ API на странице [console.scaleway.com](https://console.scaleway.com). +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` ->**💡 Ultimate Free Stack (11 провайдеров, 0 долларов США навсегда):** +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). + +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Киро (кр/) → Клод Сонет/Haiku UNLIMITED -> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 миллионов токенов в день 🔥 -> Опыления (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — ключ не нужен -> Qwen (qw/) → модели qwen3-кодеров НЕОГРАНИЧЕННО -> Gemini (gemini/) → Gemini 2.5 Flash — 1500 запросов в день бесплатно -> Cloudflare AI (см./) → 50+ моделей — 10 тыс. нейронов в день -> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1 миллион бесплатных жетонов (ЕС) -> Грок (groq/) → Лама/Джемма — 14,4 тыс. запросов в день сверхбыстро -> NVIDIA NIM (nvidia/) → 70+ открытых моделей — 40 об/мин навсегда -> Церебрас (cerebras/) → Лама/Квен самый быстрый в мире — 1 млн ток/день -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Транскрибируйте любое аудио/видео за**0 долларов США**— Deepgram лидирует с бесплатными 200 долларами, резервным вариантом AssemblyAI в 50 долларов и Groq Whisper в качестве неограниченного резервного копирования на случай чрезвычайной ситуации. +## 🎙️ Free Transcription Combo -| Провайдер | Бесплатные кредиты | Лучшая модель | Ограничение ставки | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. + +| Provider | Free Credits | Best Model | Rate Limit | | ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | -| 🟢**Дипграмм**|**200 долларов США бесплатно**(регистрация) | `nova-3` — лучшая точность, более 30 языков | Нет лимита RPM на бесплатных кредитах | -| 🔵**АИ сборки**|**50 долларов США бесплатно**(регистрация) | `universal-3-pro` — главы, настроения, персональные данные | Нет лимита RPM на бесплатных кредитах | -| 🔴**Грок**|**Бесплатно навсегда**| `whisper-large-v3` — OpenAI Whisper | 30 об/мин (скорость ограничена) | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | -**Рекомендуемая комбинация в `/dashboard/combos`:**``` +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Затем в `/dashboard/media` → вкладка**Транскрипция**: загрузите любой аудио- или видеофайл → выберите конечную точку комбо → получите транскрипцию в поддерживаемых форматах.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 создан как операционная платформа, а не просто прокси-сервер.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Особенность | Что он делает | -| ---------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------ | -| ⚡**Быстрая семья Грока-4** | Модели xAI по цене 0,20/0,50 доллара США в месяц — измеренное время 1143 мс (на 30 % быстрее, чем у Gemini 2.5 Flash) | -| 🧠**GLM-5 через Z.AI** | Выходной контекст 128 тыс., $0,5/1 млн — новейший флагман семейства GLM | -| 🔮**МиниМакс М2,5** | Рассуждение + агентские задачи по цене 0,30 долл. США/1 миллион — значительное обновление по сравнению с M2.1 | -| 🎯**Флаг вызова инструмента для каждой модели** | `toolCalling: true/false` для каждой модели в реестре — AutoCombo пропускает модели, не поддерживающие инструменты | -| 🌍**Многоязычное обнаружение намерений** | Ключевые слова PT/ZH/ES/AR в оценке AutoCombo — лучший выбор модели для неанглоязычного контента | -| 📊**Резервные варианты на основе эталонного тестирования** | Реальная задержка p95 от запросов в реальном времени обеспечивает комбинированную оценку — AutoCombo учится на реальных данных | -| 🔁**Запросить дедупликацию** | Окно дедупликации на основе хэша контента — мультиагентная безопасность, предотвращает дублирование платежей | -| 🔌**Стратегия подключаемого маршрутизатора** | Расширяемый интерфейс RouterStrategy — добавляйте собственную логику маршрутизации в виде плагинов | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Особенность | Что он делает | -| ------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------- | -| 🎮**Игровая площадка** | Страница информационной панели для непосредственного тестирования любой модели — селекторы поставщика/модели/конечной точки, редактор Monaco, потоковая передача, прерывание, синхронизация | -| 🔏**Сопоставление отпечатков пальцев CLI** | Порядок заголовков/тела для каждого провайдера в соответствии с собственными подписями CLI — переключайте для каждого провайдера в «Настройки» > «Безопасность».**Ваш IP-адрес прокси-сервера сохраняется** | -| 🤝**Поддержка ACP (протокол агента-клиента)** | Обнаружение агентов CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw и еще 9), генератор процессов, конечная точка `/api/acp/agents` | -| 🤖**Панель управления агентами ACP** | Страница «Отладка» › «Агенты» — сетка из 14 агентов со статусом установки, версией, настраиваемой формой агента для любого инструмента CLI. Пользователи**OpenCode**получают кнопку «Загрузить opencode.json», которая автоматически генерирует готовую конфигурацию для всех доступных моделей. | -| 🔧**Маршрутизация пользовательской модели `apiFormat`** | Пользовательские модели с `apiFormat: "responses"` теперь правильно перенаправляются к переводчику API ответов | -| 🏢**Изоляция рабочего пространства Кодекса** | Несколько рабочих пространств Кодекса для одного электронного письма — OAuth правильно разделяет соединения по идентификатору рабочего пространства | -| 🔄**Электронное автоматическое обновление** | Настольное приложение проверяет наличие обновлений + автоматическая установка при перезапуске | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Особенность | Что он делает | -| ----------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**Сервер MCP (25 инструментов)** | Инструменты IDE/агента через 3 транспорта: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 ядер + 3 памяти + 4 инструмента навыков | -| 🤝**Сервер A2A (JSON-RPC + SSE)** | Выполнение задач между агентами с синхронизацией и потоковой передачей | -| 🧭**Объединенная страница конечных точек** | Страница управления с вкладками Endpoint Proxy, MCP, A2A и API Endpoints | -| 🎚️**Переключатели включения/отключения службы** | Переключатели ВКЛ/ВЫКЛ для MCP и A2A с сохранением настроек (по умолчанию: ВЫКЛ) | -| 🛰️**Сердцебиение среды выполнения MCP** | Реальный статус процесса (pid, время безотказной работы, возраст контрольного сигнала, транспорт, режим области) | -| 📋**Аудитский журнал MCP** | Фильтруемые журналы аудита с указанием успехов/неуспехов и присвоением ключей | -| 🔐**Обеспечение соблюдения требований MCP** | 10 расширенных разрешений для контролируемого доступа к инструментам | -| 📡**Управление жизненным циклом задач A2A** | Список/фильтрация задач, проверка событий/артефактов, отмена запущенных задач | -| 📋**Обнаружение карты агента** | `/.well-known/agent.json` для автоматического обнаружения клиентов | -| 🧪**Тестовый жгут протокола E2E** | Реальные потоки клиента MCP SDK + A2A в `test:protocols:e2e` | -| ⚙️**Оперативный контроль** | Комбинированное переключение, применение профилей устойчивости, сброс прерывателей с одной панели управления | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Особенность | Что он делает | -| --------------------------------------------- | ------------------------------------------------------------------------------------------------ | ----------------------- | -| 🎯**Умный 4-уровневый резерв** | Авто-маршрутизация: Подписка → Ключ API → Дешево → Бесплатно | -| 📊**Отслеживание квот в реальном времени** | Подсчет токенов в реальном времени + сброс обратного отсчета для каждого провайдера | -| 🔄**Перевод формата** | OpenAI ↔ Клод ↔ Gemini ↔ Ответы с безопасными для схемы преобразованиями | -| 👥**Поддержка нескольких аккаунтов** | Несколько учетных записей у каждого провайдера с интеллектуальным выбором | -| 🔄**Автоматическое обновление токена** | Токены OAuth автоматически обновляются при повторной попытке | -| 🎨**Пользовательские комбинации** | 9 стратегий балансировки + резервный контроль цепочки | -| 🌐**Маршрутизатор с подстановочными знаками** | `provider/*` динамическая маршрутизация | -| 🧠**Продуманный бюджетный контроль** | Ограничения сквозного, автоматического, пользовательского и адаптивного рассуждения | -| 🔀**Псевдонимы моделей** | Встроенное + пользовательское сглаживание моделей и безопасность миграции | -| ⚡**Фоновая деградация** | Перенаправить фоновые задачи с низким приоритетом на более дешевые модели | -| 🧪**Умная маршрутизация с учетом задач** | Автоматический выбор модели по типу контента (кодирование/видение/анализ/суммирование) | -| 🔄**Рабочие процессы агента A2A** | Детерминированный оркестратор FSM для многоэтапного выполнения агентов с отслеживанием состояния | -| 🔀**Адаптивная маршрутизация** | Динамическое переопределение стратегии в зависимости от объема токенов и сложности запросов | -| 🎲**Разнообразие поставщиков** | Оценка энтропии Шеннона, балансирующая автоматическое комбинированное распределение трафика | -| 💬**Внедрение системных подсказок** | Глобальный контроль поведения применяется последовательно | -| 📄**Совместимость API ответов** | Полная поддержка `/v1/responses` для Кодекса и расширенных агентских рабочих процессов | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Особенность | Что он делает | -| ---------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**Генерация изображений** | `/v1/images/generations` с облачными и локальными серверными модулями | -| 📐**Встраивания** | `/v1/embeddings` для конвейеров поиска и RAG | -| 🎤**Аудиотранскрипция** | `/v1/audio/transcriptions` — 7 провайдеров (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), автоматическое определение языка, поддержка MP4/MP3/WAV | -| 🔊**Преобразование текста в речь** | `/v1/audio/speech` — 10 провайдеров (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) с правильными сообщениями об ошибках | -| 🎬**Создание видео** | `/v1/videos/generations` (рабочие процессы ComfyUI + SD WebUI) | -| 🎵**Генерация музыки** | `/v1/music/generations` (рабочие процессы ComfyUI) | -| 🛡️**Модерация** | `/v1/moderations` проверки безопасности | -| 🔀**Изменение рейтинга** | `/v1/rerank` для оценки релевантности | -| 🔍**Веб-поиск**🆕 | `/v1/search` — 5 провайдеров (Serper, Brave, Perplexity, Exa, Tavily), более 6500 бесплатно в месяц, автоматическое переключение при отказе, кэш | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Особенность | Что он делает | -| ----------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**Автоматические выключатели** | Отключение/восстановление для каждой модели с пороговым управлением | -| 🎯**Модели с поддержкой конечных точек** | Пользовательские модели объявляют поддерживаемые конечные точки + формат API | -| 🛡️**Антигромовое стадо** | Защита мьютексом и семафором при повторных попытках/частотах событий | -| 🧠**Семантический кэш + сигнатуры** | Снижение затрат и задержек за счет двух слоев кэша | -| ⚡**Запросить идемпотентность** | Окно защиты от дубликатов | -| 🔒**Подмена отпечатка пальца TLS** | Браузерный отпечаток TLS —**уменьшает обнаружение ботов и пометку учетных записей** | -| 🔏**Сопоставление отпечатков пальцев CLI** | Соответствует собственным сигнатурам запросов CLI —**снижает риск бана, сохраняя при этом IP-адрес прокси** | -| 🌐**IP-фильтрация** | Управление списками разрешенных/черных списков для открытых развертываний | -| 📊**Редактируемые ограничения ставок** | Настраиваемые глобальные ограничения/пределы на уровне поставщика с сохранением | -| 📉**Изящная деградация** | Многоуровневые возможности резервного копирования, защищающие операции основного шлюза | -| 📜**След аудита конфигурации** | Отслеживание изменений на основе различий, предотвращающее дрейф в работе с помощью простых откатов | -| ⏳**Синхронизация состояния поставщика** | Проактивный мониторинг истечения срока действия токена, вызывающий оповещения перед сбоями авторизации | -| 🚪**Автоматическое отключение заблокированных аккаунтов** | Операционный выключатель автоматически запечатывает постоянно заблокированные токены-аккаунты | -| 🔑**Управление ключами API + определение области действия** | Безопасная выдача/ротация ключей и контроль модели/поставщика | -| 👁️**Раскрытие ключа API с ограниченной областью**🆕 | Согласие на восстановление ключей API через `ALLOW_API_KEY_REVEAL` | -| 🛡️**Защищенные `/модели`** | Дополнительная аутентификация и скрытие поставщика для каталога моделей | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Особенность | Что он делает | -| -------------------------------------------- | -------------------------------------------------------------------------------------- | ---------------------------- | -| 📝**Запрос + регистрация через прокси** | Полное ведение журнала запросов/ответов и прокси | -| 📉**Подробные журналы в потоковом режиме**🆕 | Четко реконструирует потоки полезной нагрузки SSE в пользовательский интерфейс | -| 📋**Единая панель журналов** | Представления запросов, прокси, аудита и консоли на одной странице | -| 🔍**Запросить телеметрию** | Задержка p50/p95/p99 и отслеживание запросов | -| 🏥**Панель здоровья** | Время работы, состояния прерывателей, блокировки, статистика кэша | -| 💰**Отслеживание затрат** | Контроль бюджета и прозрачность цен для каждой модели | -| 📈**Аналитическая визуализация** | Анализ использования моделей/провайдеров и обзор тенденций | -| 🧪**Система оценки** | Тестирование золотого сета с настраиваемыми стратегиями сопоставления | -| 📡**Живая диагностика**🆕 | Обход семантического кэша для точного комбинированного тестирования в реальном времени | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Особенность | Что он делает | -| ------------------------------------------- | ------------------------------------------------------------------------------------------- | --------------------- | -| 🌐**Развертывание где угодно** | Локальный хост, VPS, Docker, облачные среды | -| 🚇**Туннель Cloudflare**🆕 | Интеграция Quick Tunnel в один клик с панели управления | -| 🔑**Фильтрация модели ключей API** | Собственный ответ /v1/models фильтруется с помощью назначенных ролей контекста носителя | -| ⚡**Умный обход кэша** | Настраиваемая эвристика TTL и элементы управления принудительной выборкой | -| 🔄**Резервное копирование/восстановление** | Экспорт/импорт и потоки аварийного восстановления | -| 🧙**Мастер адаптации** | Пошаговая настройка при первом запуске | -| 🔧**Панель инструментов CLI** | Настройка популярных инструментов кодирования в один клик | -| 🎮**Игровая площадка** | Протестируйте любого поставщика/модель/конечную точку с помощью панели управления | -| 🔏**Переключатель отпечатков пальцев CLI** | Сопоставление отпечатков пальцев для каждого поставщика в меню «Настройки» > «Безопасность» | -| 🌐**i18n (30 языков)** | Полная информационная панель + языковая поддержка документации с RTL | -| 🧹**Очистить все модели** | Очистка списка моделей в сведениях о поставщике в один клик | -| 👁️**Элементы управления боковой панелью**🆕 | Скрыть компоненты и интеграции в настройках внешнего вида | -| 📋**Шаблоны задач** | Стандартизированные шаблоны GitHub для ошибок и функций | -| 📂**Каталог пользовательских данных** | `DATA_DIR` переопределить место хранения | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1294,105 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -При сбое квоты, скорости или работоспособности OmniRoute автоматически переходит к следующему кандидату без ручного переключения.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A доступны для обнаружения в пользовательском интерфейсе и документах (не скрыты) -- API-интерфейсы состояния протокола предоставляют оперативные данные в реальном времени (`/api/mcp/*`, `/api/a2a/*`) -- Панели мониторинга включают действия для операций второго дня (переключение комбо, сброс выключателя, отмена задачи)#### Translator + validation workflow +#### Protocol management that is visible and operable -Раздел «Переводчик» включает в себя: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Игровая площадка**: запросить проверку трансформации. -**Тестер чата**: полный цикл запросов и ответов. -**Тестовый стенд**: несколько случаев за один запуск. -**Живой монитор**: просмотр трафика в реальном времени. +#### Translator + validation workflow -Плюс проверка протокола на реальных клиентах с помощью `npm run test:protocols:e2e`. +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Справочник по инструментам, конфигурации IDE и примеры клиентов. +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[README сервера A2A](src/lib/a2a/README.md)**— навыки, методы JSON-RPC, потоковая передача и жизненный цикл задачи.## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute включает в себя встроенную систему оценки для проверки качества ответов LLM на соответствие «золотому набору». Доступ к нему осуществляется через**Аналитика → Оценки**на панели управления.### Built-in Golden Set +## 🧪 Evaluations (Evals) -Предварительно загруженный «Золотой набор OmniRoute» содержит тестовые сценарии для: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Приветствую, математика, география, генерация кода -- Соответствие формата JSON, перевод, создание уценки -- Отказ безопасности (вредный контент), подсчет, булева логика### Evaluation Strategies +### Built-in Golden Set -| Стратегия | Описание | Пример | -| ---------------------- | ---------------------------------------------------------- | ----------------------------- | --- | -| `точный` | Вывод должен точно совпадать | `"4"` | -| `содержит` | Вывод должен содержать подстроку (без учета регистра) | `"Париж"` | -| `регулярное выражение` | Вывод должен соответствовать шаблону регулярного выражения | `"1.*2.*3"` | -| `обычай` | Пользовательская функция JS возвращает true/false | `(выход) => выход.длина > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) -<подробности> +
+🧩 MCP Setup (Model Context Protocol) -🧩 Настройка MCP (протокол контекста модели) +Start MCP transport in stdio mode: -Запустите транспорт MCP в режиме stdio:```bash +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Рекомендуемый порядок проверки: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Подключите клиент MCP через stdio. -2. Запустите omniroute_get_health. -3. Запустите `omniroute_list_combos`. -4. Откройте `/dashboard/mcp`, чтобы проверить пульс, активность и аудит. - -Полезные API для автоматизации: +Useful APIs for automation: - `GET /api/mcp/status` - `GET /api/mcp/tools` - `GET /api/mcp/audit` -- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats` -<подробности> -🤝 Настройка A2A (Agent2Agent) + -Откройте для себя агента:```bash +
+🤝 A2A Setup (Agent2Agent) + +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Отправить задачу:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` - -Управлять жизненным циклом: +Manage lifecycle: - `GET /api/a2a/status` - `GET /api/a2a/tasks` - `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -Оперативный интерфейс: +Operational UI: -- `/dashboard/a2a` для наблюдения за задачей/состоянием/потоком и действий по дыму
+- `/dashboard/a2a` for task/state/stream observability and smoke actions -<подробности> -🧪 Сквозная проверка протокола + -Проверьте оба протокола на реальных клиентах:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Это проверяет: +This verifies: -- Клиент MCP SDK подключается/списывает/вызовает -- Обнаружение/отправка/потоковая передача A2A/получение/отмена -- Перекрестная проверка данных в аудите MCP и API управления задачами A2A.
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs -<подробности> + -💳 Поставщики подписки### Claude Code (Pro/Max) +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1405,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Совет для профессионалов.**Используйте Opus для сложных задач и Sonnet для скорости. OmniRoute отслеживает квоту на каждую модель!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1419,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Каждая учетная запись Кодекса теперь имеет переключатели политики в «Панель управления -> Поставщики»: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (ВКЛ/ВЫКЛ): применить политику порогового значения 5-часового окна. -- «Еженедельно» (ВКЛ/ВЫКЛ): применение политики порогового значения еженедельного окна. -- Пороговое поведение: когда включенное окно достигает >=90% использования, эта учетная запись пропускается. -- Порядок ротации: OmniRoute автоматически направляет данные к следующей подходящей учетной записи Кодекса. -- Поведение при сбросе: по истечении времени, установленного поставщиком `resetAt`, учетная запись снова становится подходящей для автоматически. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Сценарии: +Scenarios: -- `5h ON` + `Weekly ON`: учетная запись пропускается, когда любое окно достигает порогового значения. -- «5 часов ВЫКЛ» + «Еженедельно ВКЛ»: только еженедельное использование может заблокировать учетную запись. -- «5 часов включено» + «Еженедельно выключено»: только 5-часовое использование может заблокировать учетную запись. -- `resetAt` пройден: учетная запись снова входит в ротацию автоматически (без повторного включения вручную).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1444,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Лучшая цена:**Огромный уровень бесплатного пользования! Используйте это перед платными уровнями.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1459,74 +1662,91 @@ Models:
-<подробности> +
+🔑 API Key Providers -🔑 Поставщики ключей API### NVIDIA NIM (FREE developer access — 70+ models) +### NVIDIA NIM (FREE developer access — 70+ models) -1. Зарегистрируйтесь: [build.nvidia.com](https://build.nvidia.com) -2. Получите бесплатный ключ API (включая 1000 кредитов вывода) -3. Панель управления → Добавить провайдера → NVIDIA NIM: - - Ключ API: `nvapi-your-key` +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Модели:**nvidia/llama-3.3-70b-instruct, nvidia/mistral-7b-instruct и более 50 других. +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -**Совет для профессионалов.**API-интерфейс, совместимый с OpenAI, отлично работает с преобразованием форматов OmniRoute!### DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -1. Зарегистрируйтесь: [platform.deepseek.com](https://platform.deepseek.com) -2. Получите ключ API -3. Панель управления → Добавить провайдера → DeepSeek. +### DeepSeek -**Модели:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -1. Зарегистрируйтесь: [console.groq.com](https://console.groq.com) -2. Получите ключ API (включая бесплатный уровень) -3. Панель управления → Добавить провайдера → Groq. +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Модели:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +### Groq (Free Tier Available!) -**Совет для профессионалов.**Сверхбыстрый вывод — лучший вариант для кодирования в реальном времени!### OpenRouter (100+ Models) +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -1. Зарегистрируйтесь: [openrouter.ai](https://openrouter.ai) -2. Получите ключ API -3. Панель управления → Добавить провайдера → OpenRouter. +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Модели.**Получите доступ к более чем 100 моделям от всех основных поставщиков с помощью одного ключа API. +**Pro Tip:** Ultra-fast inference — best for real-time coding! -**Поведение информационной панели:**Управление моделями OpenRouter осуществляется из раздела**Доступные модели**. Добавление вручную, импорт и автоматическая синхронизация обновляют один и тот же список.
+### OpenRouter (100+ Models) -<подробности> +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -💰 Дешевые провайдеры (резервный)### GLM-4.7 (Daily reset, $0.6/1M) +**Models:** Access 100+ models from all major providers through a single API key. -1. Зарегистрируйтесь: [Zhipu AI](https://open.bigmodel.cn/) -2. Получите ключ API из плана кодирования. -3. Панель управления → Добавить ключ API: - - Провайдер: `glm` - - Ключ API: `ваш-ключ` +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -**Используйте:**`glm/glm-4.7` + -**Совет для профессионалов.**План кодирования предлагает 3-кратную квоту за 1/7 стоимости! Сброс ежедневно в 10:00.### MiniMax M2.1 (5h reset, $0.20/1M) +
+💰 Cheap Providers (Backup) -1. Зарегистрируйтесь: [MiniMax](https://www.minimax.io/) -2. Получите ключ API -3. Панель управления → Добавить ключ API. +### GLM-4.7 (Daily reset, $0.6/1M) -**Используйте:**`minimax/MiniMax-M2.1` +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**Совет для профессионалов:**Самый дешевый вариант для длинного контекста (1 млн токенов)!### Kimi K2 ($9/month flat) +**Use:** `glm/glm-4.7` -1. Подпишитесь: [Moonshot AI](https://platform.moonshot.ai/) -2. Получите ключ API -3. Панель управления → Добавить ключ API. +**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. -**Используйте:**`kimi/kimi-latest` +### MiniMax M2.1 (5h reset, $0.20/1M) -**Совет для профессионалов:**Фиксированные 9 долларов США в месяц за 10 миллионов токенов = эффективная стоимость 0,90 долларов США/1 миллион долларов!
+1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key -<подробности> +**Use:** `minimax/MiniMax-M2.1` -🆓 БЕСПЛАТНЫЕ поставщики (аварийное резервное копирование)### Qoder (5 FREE models via OAuth) +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1567,9 +1787,10 @@ Models:
-<подробности> +
+🎨 Create Combos -🎨 Создавайте комбинации### Example 1: Maximize Subscription → Cheap Backup +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1597,9 +1818,10 @@ Cost: $0 forever!
-<подробности> +
+🔧 CLI Integration -🔧 Интеграция CLI### Cursor IDE +### Cursor IDE ``` Settings → Models → Advanced: @@ -1610,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Используйте страницу**Инструменты CLI**на панели управления для настройки одним щелчком мыши или отредактируйте `~/.claude/settings.json` вручную.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1621,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Вариант 1 – Панель управления (рекомендуется):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Вариант 2 — Вручную:**Отредактируйте `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1638,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Примечание.**OpenClaw работает только с локальным OmniRoute. Используйте «127.0.0.1» вместо «localhost», чтобы избежать проблем с разрешением IPv6.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1652,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Шаг 1.**Добавьте OmniRoute в качестве собственного поставщика:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Шаг 2.**Создайте/измените `opencode.json` в корне вашего проекта:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1678,117 +1909,130 @@ opencode } } } -```` +``` -**Шаг 3:**Выберите модель в OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Совет.**Добавьте любую модель, доступную в конечной точке `/v1/models` OmniRoute, в раздел `models`. Используйте формат «поставщик/идентификатор модели» на панели управления OmniRoute.
+ --- ## Устранение неполадок -<подробности> -Нажмите, чтобы развернуть руководство по устранению неполадок +
+Click to expand troubleshooting guide -**"Языковая модель не предоставила сообщения"** +**"Language model did not provide messages"** -- Квота провайдера исчерпана → Проверьте трекер квот на панели управления. -- Решение: используйте запасной вариант комбо или переключитесь на более дешевый уровень. +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**Ограничение скорости** +**Rate limiting** -- Исчерпана квота подписки → Переход на GLM/MiniMax -- Добавить комбо: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**Срок действия токена OAuth истек** +**OAuth token expired** -- Автоматическое обновление OmniRoute -- Если проблемы не устранены: Панель управления → Поставщик → Переподключиться. +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Высокие затраты** +**High costs** -- Проверьте статистику использования в Личном кабинете → Расходы. -- Переключение основной модели на GLM/MiniMax. -- Используйте уровень бесплатного пользования (Gemini CLI, Qoder) для некритических задач. +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Порты панели управления/API неверны** +**Dashboard/API ports are wrong** -- `PORT` — это канонический базовый порт (и порт API по умолчанию). -- API_PORT переопределяет только прослушиватель API, совместимый с OpenAI. -- DASHBOARD_PORT переопределяет только панель мониторинга/прослушиватель Next.js. -– Установите `NEXT_PUBLIC_BASE_URL` для вашей панели управления/публичного URL-адреса (для обратных вызовов OAuth). +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Ошибки синхронизации с облаком** +**Cloud sync errors** -– Убедитесь, что `BASE_URL` указывает на ваш работающий экземпляр. -– Убедитесь, что `CLOUD_URL` указывает на ожидаемую конечную точку облака. -- Сохраняйте значения `NEXT_PUBLIC_*` в соответствии со значениями на стороне сервера. +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Первый вход в систему не работает** +**First login not working** -- Проверьте `INITIAL_PASSWORD` в `.env` -- Если параметр не установлен, резервный пароль — «123456». +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Нет журналов запросов** +**No request logs** -- Артефакты запроса записываются в `DATA_DIR/call_logs/` как один файл JSON для каждого запроса. -- Включите захват конвейера на панели мониторинга → Журналы → Журналы запросов, если вам нужны подробные полезные данные для каждого этапа. -- Установите `APP_LOG_TO_FILE=true`, если вы также хотите, чтобы журналы консоли приложения находились в `logs/application/app.log` -- При необходимости настройте `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` и `CALL_LOG_MAX_ENTRIES`. +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Проверка соединения показывает «Недействительно» для провайдеров, совместимых с OpenAI** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Многие провайдеры не предоставляют конечную точку `/models`. -- OmniRoute v1.0.6+ включает резервную проверку посредством завершения чата. -– Убедитесь, что базовый URL-адрес содержит суффикс `/v1`.### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ Важно для пользователей, использующих OmniRoute на VPS, Docker или любом удаленном сервере**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -Поставщики**Antigravity**и**Gemini CLI**используют**Google OAuth 2.0**. Google требует, чтобы redirect_uri в потоке OAuth точно соответствовал одному из предварительно зарегистрированных URI в Google Cloud Console приложения. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -Учетные данные OAuth, включенные в OmniRoute, регистрируются**только для «localhost»**. Когда вы получаете доступ к OmniRoute на удаленном сервере (например, https://omniroute.myserver.com), Google отклоняет аутентификацию с помощью:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Вам необходимо создать**Идентификатор клиента OAuth 2.0**в Google Cloud Console с URI вашего сервера.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Откройте Google Cloud Console** +#### Step-by-step -Перейдите по адресу: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials). +**1. Open Google Cloud Console** -**2. Создайте новый идентификатор клиента OAuth 2.0** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Нажмите**"+ Создать учетные данные"**→**"Идентификатор клиента OAuth"**. -- Тип приложения:**"Веб-приложение"** -- Имя: любое, что вам нравится (например, «OmniRoute Remote») +**2. Create a new OAuth 2.0 Client ID** -**3. Добавить авторизованные URI перенаправления** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -В поле**"Авторизованные URI перенаправления"**добавьте:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Замените `your-server.com` на домен или IP-адрес вашего сервера (при необходимости укажите порт, например `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. Сохраните и скопируйте учетные данные** +After creating, Google will show the **Client ID** and **Client Secret**. -После создания Google покажет**Идентификатор клиента**и**Секрет клиента**. +**5. Set environment variables** -**5. Установить переменные среды** +In your `.env` (or Docker environment variables): -В вашем `.env` (или переменных среды Docker):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1797,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Перезапустите OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Попробуйте подключиться еще раз** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Панель управления → Провайдеры → Антигравитация (или Gemini CLI) → OAuth. +Google will now redirect correctly to `https://your-server.com/callback`. -Google теперь будет правильно перенаправляться на https://your-server.com/callback.--- +--- #### Temporary workaround (without custom credentials) -Если вы не хотите настраивать свои собственные учетные данные прямо сейчас, вы все равно можете использовать**ручной поток URL**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute открывает URL-адрес авторизации Google. -2. После авторизации Google пытается перенаправить на `localhost` (на удаленном сервере не получается) -3.**Скопируйте полный URL-адрес**из адресной строки браузера (даже если страница не загружается). -4. Вставьте этот URL-адрес в поле, отображаемое в модальном окне подключения OmniRoute. -5. Нажмите**"Подключиться"**. +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Это работает, поскольку код авторизации в URL-адресе действителен независимо от того, загружена ли страница перенаправления.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. -<подробности> -🇧🇷 Версия на португальском языке#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Используйте**Антигравитацию**и**Gemini CLI**, используя**Google OAuth 2.0**для аутентификации. Google объяснил, что `redirect_uri` не использует поток OAuth, поэтому**exatamente**указывает на URI, предварительно подготовленные к кадастрам, без приложения Google Cloud Console. +
+🇧🇷 Versão em Português -В качестве подтверждения OAuth не используется OmniRoute, который является кадастровым**apenas para `localhost`**. Когда вы хотите получить доступ к OmniRoute на удаленном сервере (например: «https://omniroute.meuservidor.com») или Google подтвердить аутентификацию через:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Точно назовите**Идентификатор клиента OAuth 2.0**без Google Cloud Console с URI вашего сервера.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Доступ к Google Cloud Console** +#### Passo a passo -Абра: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Acesse o Google Cloud Console** -**2. Crie um novo Идентификатор клиента OAuth 2.0** +Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Нажимаем**"+ Create Credentials"**→**"Идентификатор клиента OAuth"** - – Тип приложения:**"Веб-приложение"**. -- Имя: имя escolha qualquer (например: OmniRoute Remote). +**2. Crie um novo OAuth 2.0 Client ID** -**3. Adicione как авторизованные URI перенаправления** +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -Нет надписи**"Авторизованные URI перенаправления"**, дополнение:``` +**3. Adicione as Authorized Redirect URIs** + +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Замените `seu-servidor.com` на домен или IP-адрес своего сервера (включая необходимый порт, например: `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. Сохраните и скопируйте как удостоверение** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -В ответ на запрос Google выберите**Идентификатор клиента**и**Секрет клиента**. +**5. Configure as variáveis de ambiente** -**5. Настроить как вариант окружения** +No seu `.env` (ou nas variáveis de ambiente do Docker): -Нет файла .env (или вариантов окружения Docker):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1876,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Рейниси или OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute - -```` +``` **7. Tente conectar novamente** -Панель управления → Провайдеры → Антигравитация (или Gemini CLI) → OAuth +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Агора или Google перенаправляются на «https://seu-servidor.com/callback» и обеспечивают функцию аутентификации.--- +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. + +--- #### Workaround temporário (sem configurar credenciais próprias) -Если вы не хотите писать собственные данные, вы можете использовать или Fluxo**руководство по URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. OmniRoute перейдет по URL-адресу авторизации Google. -2. После вашего авторизации вы можете перенаправить Google на «localhost» (если нет удаленного сервера) -3.**Скопируйте URL-адрес полностью**из окна браузера (сообщение, которое страница не поддерживает) -4. URL-адрес не отображается в модальном режиме подключения к OmniRoute. -5. Нажимаем**"Подключиться"** +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) +4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute +5. Clique em **"Connect"** -> Эта функция обхода позволяет выполнить перенаправление на URL-адрес независимо от кода авторизации или перенаправления.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1914,64 +2171,73 @@ docker restart omniroute ## 🛠️ Tech Stack -<подробности> -Нажмите, чтобы развернуть подробную информацию о технологическом стеке +
+Click to expand tech stack details --**Среда выполнения**: Node.js 18–22 LTS (⚠️ Node.js 24+**не поддерживается**— собственные двоичные файлы `better-sqlite3` несовместимы) --**Язык**: TypeScript 5.9 —**100% TypeScript**в `src/` и `open-sse/` (ноль `any` в основных модулях, начиная с версии 2.0) --**Фреймворк**: Next.js 16 + React 19 + Tailwind CSS 4. --**База данных**: LowDB (JSON) + SQLite (состояние домена + журналы прокси-сервера + аудит MCP + решения о маршрутизации). --**Схемы**: Zod (проверка ввода-вывода инструмента MCP, контракты API) --**Протоколы**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE). --**Потоковая передача**: события, отправленные сервером (SSE). --**Аутентификация**: OAuth 2.0 (PKCE) + JWT + ключи API + авторизация с ограниченной областью действия MCP. --**Тестирование**: средство запуска тестов Node.js + Vitest (более 900 тестов, включая модульные тесты, интеграцию, E2E) --**CI/CD**: действия GitHub (автоматическая публикация npm + Docker Hub при выпуске) --**Веб-сайт**: [omniroute.online](https://omniroute.online) --**Пакет**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Докер**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Устойчивость**: автоматический выключатель, экспоненциальное отключение, защита от громового стада, подмена TLS, автоматическое комбинированное самовосстановление.
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Документация -| Документ | Описание | -| --------------------------------------------- | -------------------------------------------------- | -| [Руководство пользователя](docs/USER_GUIDE.md) | Провайдеры, комбинации, интеграция CLI, развертывание | -| [Справочник API](docs/API_REFERENCE.md) | Все конечные точки с примерами | -| [Сервер MCP](open-sse/mcp-server/README.md) | 16 инструментов MCP, конфигурации IDE, клиенты Python/TS/Go | -| [Сервер A2A](src/lib/a2a/README.md) | Протокол JSON-RPC 2.0, навыки, потоковая передача, управление задачами | -| [Механизм автоматического комбинирования](docs/auto-combo.md) | 6-факторная оценка, пакеты режимов, самовосстановление | -| [Устранение неполадок](docs/TROUBLESHOOTING.md) | Распространенные проблемы и решения | -| [Архитектура](docs/ARCHITECTURE.md) | Архитектура и внутреннее устройство системы | -| [Содействие](CONTRIBUTING.md) | Настройка и рекомендации по разработке | -| [Спецификация OpenAPI](docs/openapi.yaml) | Спецификация OpenAPI 3.0 | -| [Политика безопасности](SECURITY.md) | Отчеты об уязвимостях и методы обеспечения безопасности | -| [Развертывание виртуальной машины](docs/VM_DEPLOYMENT_GUIDE.md) | Полное руководство: настройка VM + nginx + Cloudflare | -| [Галерея функций](docs/FEATURES.md) | Визуальный обзор панели управления со скриншотами | -| [Контрольный список выпуска](docs/RELEASE_CHECKLIST.md) | Этапы проверки перед выпуском |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -В OmniRoute запланировано**более 210 функций**на нескольких этапах разработки. Вот ключевые направления: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Категория | Планируемые функции | Основные моменты | +| Category | Planned Features | Highlights | | ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | -| 🧠**Маршрутизация и разведка**| 25+ | Маршрутизация с минимальной задержкой, маршрутизация на основе тегов, предварительная проверка квот, выбор учетной записи P2C | -| 🔒**Безопасность и соответствие требованиям**| 20+ | Усиление защиты SSRF, сокрытие учетных данных, ограничение скорости на конечную точку, определение области действия ключей управления | -| 📊**Наблюдаемость**| 15+ | Интеграция OpenTelemetry, мониторинг квот в реальном времени, отслеживание затрат на каждую модель | -| 🔄**Интеграция поставщиков**| 20+ | Реестр динамических моделей, время восстановления поставщика, Кодекс с несколькими учетными записями, анализ квот Copilot | -| ⚡**Представление**| 15+ | Двойной уровень кэша, кэш запросов, кэш ответов, потоковая поддержка активности, пакетный API | -| 🌐**Экосистема**| 10+ | WebSocket API, горячая перезагрузка конфигурации, распределенное хранилище конфигураций, коммерческий режим |### 🔜 Coming Soon +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**Интеграция OpenCode**— поддержка встроенного поставщика для IDE OpenCode AI для кодирования. -- 🔗**Интеграция TRAE**— Полная поддержка среды разработки TRAE AI. -- 📦**Пакетный API**— асинхронная пакетная обработка массовых запросов. -- 🎯**Маршрутизация на основе тегов**— маршрутизация запросов на основе пользовательских тегов и метаданных. -- 💰**Стратегия наименьших затрат**— автоматически выбирает самого дешевого доступного провайдера. +### 🔜 Coming Soon -> 📝 Полные спецификации функций доступны в [`docs/new-features/`](docs/new-features/) (217 подробных спецификаций)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1979,18 +2245,20 @@ docker restart omniroute ### How to Contribute -1. Форкните репозиторий -2. Создайте свою ветку функций (`git checkout -b Feature/amazing-feature`). -3. Зафиксируйте изменения (`git commit -m 'Добавить замечательную функцию'`) -4. Нажмите на ветку (`git push origin Feature/amazing-feature`) -5. Откройте запрос на включение +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -См. [CONTRIBUTING.md](CONTRIBUTING.md) для получения подробных инструкций.### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -2002,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Особая благодарность**[9router](https://github.com/decolua/9router)**от**[decolua](https://github.com/decolua)**— оригинального проекта, вдохновившего на создание этого форка. OmniRoute опирается на эту невероятную основу с дополнительными функциями, мультимодальными API и полной переработкой TypeScript. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Особая благодарность**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— оригинальной реализации Go, которая послужила вдохновением для создания этого порта JavaScript.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Лицензия -Лицензия MIT — подробности см. в разделе [ЛИЦЕНЗИЯ](ЛИЦЕНЗИЯ).--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/ru/docs/ARCHITECTURE.md b/docs/i18n/ru/docs/ARCHITECTURE.md index ab9cedbd05..bf593c582a 100644 --- a/docs/i18n/ru/docs/ARCHITECTURE.md +++ b/docs/i18n/ru/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Последнее обновление: 28 марта 2026 г._## Executive Summary -OmniRoute — это локальный шлюз маршрутизации AI и панель управления, созданная на основе Next.js. -Он предоставляет единую конечную точку, совместимую с OpenAI (`/v1/*`), и маршрутизирует трафик между несколькими вышестоящими поставщиками с трансляцией, резервным копированием, обновлением токена и отслеживанием использования. -Основные возможности: +_Last updated: 2026-03-28_ -- OpenAI-совместимая поверхность API для CLI/инструментов (28 поставщиков) -- Трансляция запроса/ответа в форматах провайдера. -- Резервный вариант комбо-модели (последовательность из нескольких моделей) -- Резервный вариант на уровне учетной записи (несколько учетных записей для каждого провайдера) -- Управление подключением к поставщику OAuth + API-ключей -- Генерация встраивания через `/v1/embeddings` (6 поставщиков, 9 моделей) -- Генерация изображений через `/v1/images/generations` (4 поставщика, 9 моделей) -- Подумайте о разборе тегов (`...`) для моделей рассуждений. -- Очистка ответов для строгой совместимости OpenAI SDK. -- Нормализация ролей (разработчик→система, система→пользователь) для совместимости между поставщиками. -- Преобразование структурированного вывода (json_schema → Gemini responseSchema) -- Локальное сохранение поставщиков, ключей, псевдонимов, комбинаций, настроек, цен. -- Отслеживание использования/расходов и регистрация запросов -- Дополнительная облачная синхронизация для синхронизации нескольких устройств/состояний. -- Список разрешенных/блокированных IP-адресов для контроля доступа к API. -- Продуманное управление бюджетом (сквозное/автоматическое/настраиваемое/адаптивное) -- Оперативное внедрение глобальной системы -- Отслеживание сеансов и снятие отпечатков пальцев -- Расширенное ограничение скорости для каждой учетной записи с помощью профилей для конкретного поставщика. -- Схема автоматического выключателя для устойчивости поставщика -- Анти-громовая защита стада с блокировкой мьютекса -- Кэш дедупликации запросов на основе сигнатур. -- Уровень домена: доступность модели, правила затрат, резервная политика, политика блокировки. -- Сохранение состояния домена (кэш сквозной записи SQLite для резервных копий, бюджетов, блокировок, автоматических выключателей) -- Механизм политики для централизованной оценки запросов (блокировка → бюджет → резервный вариант) -- Запрос телеметрии с агрегацией задержек p50/p95/p99. -- Идентификатор корреляции (X-Request-Id) для сквозной трассировки. -- Ведение журнала аудита соответствия с возможностью отказа для каждого ключа API. -- Система оценки для обеспечения качества LLM -- Панель управления устойчивостью пользовательского интерфейса с отображением состояния автоматического выключателя в реальном времени. -- Модульные поставщики OAuth (12 отдельных модулей в `src/lib/oauth/providers/`) +## Executive Summary -Основная модель времени выполнения: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- Маршруты приложений Next.js в `src/app/api/*` реализуют как API панели мониторинга, так и API совместимости. -- Общее ядро SSE/маршрутизации в `src/sse/*` + `open-sse/*` управляет выполнением, трансляцией, потоковой передачей, резервным копированием и использованием поставщика.## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Среда выполнения локального шлюза -- API-интерфейсы управления информационной панелью -- Аутентификация поставщика и обновление токена -- Запросить перевод и потоковую передачу SSE -- Локальное состояние + постоянство использования -- Дополнительная оркестровка облачной синхронизации.### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Реализация облачного сервиса за `NEXT_PUBLIC_CLOUD_URL` -- Соглашение об уровне обслуживания поставщика/плоскость управления вне локального процесса. -- Сами внешние двоичные файлы CLI (Claude CLI, Codex CLI и т. д.)## Dashboard Surface (Current) +### Out of Scope -Главные страницы в `src/app/(dashboard)/dashboard/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — быстрый старт + обзор провайдера -- `/dashboard/endpoint` — прокси конечной точки + MCP + A2A + вкладки конечной точки API -- `/dashboard/providers` — подключения и учетные данные провайдера. -- `/dashboard/combos` — комбинированные стратегии, шаблоны, правила маршрутизации модели. -- `/dashboard/costs` — агрегирование затрат и видимость цен. -- `/dashboard/analytics` — аналитика и оценка использования. -- `/dashboard/limits` — управление квотами/ставками -- `/dashboard/cli-tools` — подключение CLI, обнаружение во время выполнения, генерация конфигурации. -- `/dashboard/agents` — обнаруженные агенты ACP + регистрация специального агента. -- `/dashboard/media` — площадка для изображений/видео/музыки. -- `/dashboard/search-tools` — тестирование и история поискового провайдера. -- `/dashboard/health` — время безотказной работы, автоматические выключатели, ограничения скорости. -- `/dashboard/logs` — логи запроса/прокси/аудита/консоли. -- `/dashboard/settings` — вкладки настроек системы (общие, маршрутизация, комбо по умолчанию и т.д.) -- `/dashboard/api-manager` — жизненный цикл ключа API и разрешения модели.## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Основные каталоги: +Main directories: -- `src/app/api/v1/*` и `src/app/api/v1beta/*` для API совместимости. -- `src/app/api/*` для API управления/конфигурации. -- Далее перезаписывается в `next.config.mjs` карта `/v1/*` на `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Важные пути совместимости: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` — включает пользовательские модели с `custom: true` -- `src/app/api/v1/embeddings/route.ts` — генерация встраивания (6 провайдеров) -- `src/app/api/v1/images/generations/route.ts` — генерация изображений (4+ провайдера, включая Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — специальный чат для каждого провайдера. -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — выделенные встраивания для каждого провайдера. -- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — отдельные изображения для каждого провайдера. +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` - `src/app/api/v1beta/models/[...path]/route.ts` -Домены управления: +Management domains: -- Аутентификация/настройки: `src/app/api/auth/*`, `src/app/api/settings/*` -- Поставщики/соединения: `src/app/api/providers*` -- Узлы поставщика: `src/app/api/provider-nodes*` -- Пользовательские модели: `src/app/api/provider-models` (GET/POST/DELETE) -- Каталог моделей: `src/app/api/models/route.ts` (GET) -- Конфигурация прокси: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- Ключи/псевдонимы/комбо/цены: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- Использование: `src/app/api/usage/*` -- Синхронизация/облако: `src/app/api/sync/*`, `src/app/api/cloud/*` -- Помощники по инструментам CLI: `src/app/api/cli-tools/*` -- IP-фильтр: `src/app/api/settings/ip-filter` (GET/PUT) -- Бюджет мышления: `src/app/api/settings/thinking-budget` (GET/PUT) -- Системное приглашение: `src/app/api/settings/system-prompt` (GET/PUT) -- Сеансы: `src/app/api/sessions` (GET) -- Ограничения скорости: `src/app/api/rate-limits` (GET) -- Устойчивость: `src/app/api/resilience` (GET/PATCH) — профили провайдера, автоматический выключатель, состояние ограничения скорости. -- Сброс устойчивости: `src/app/api/resilience/reset` (POST) — сброс прерывателей + время восстановления. -- Статистика кэширования: `src/app/api/cache/stats` (GET/DELETE) -- Доступность модели: `src/app/api/models/availability` (GET/POST) -- Телеметрия: `src/app/api/telemetry/summary` (GET) -- Бюджет: `src/app/api/usage/budget` (GET/POST) -- Резервные цепочки: `src/app/api/fallback/chains` (GET/POST/DELETE) -- Аудит соответствия: `src/app/api/compliance/audit-log` (GET) -- Оценки: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- Политики: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -Модули основного потока: +## 2) SSE + Translation Core -- Запись: `src/sse/handlers/chat.ts` - — Базовая оркестровка: `open-sse/handlers/chatCore.ts` -- Адаптеры выполнения поставщика: `open-sse/executors/*` -- Конфигурация обнаружения формата/провайдера: `open-sse/services/provider.ts` -- Анализ/решение модели: `src/sse/services/model.ts`, `open-sse/services/model.ts` - — Логика возврата учетной записи: `open-sse/services/accountFallback.ts` -- Реестр переводов: `open-sse/translator/index.ts` -- Преобразования потока: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` - — Извлечение/нормализация использования: `open-sse/utils/usageTracking.ts` -- Анализатор тегов Think: `open-sse/utils/thinkTagParser.ts` -- Обработчик встраивания: `open-sse/handlers/embeddings.ts` -- Реестр поставщиков встраивания: `open-sse/config/embeddingRegistry.ts` - — Обработчик генерации изображения: `open-sse/handlers/imageGeneration.ts` - — Реестр поставщика изображений: `open-sse/config/imageRegistry.ts` -- Санитация ответа: `open-sse/handlers/responseSanitizer.ts` -- Нормализация ролей: `open-sse/services/roleNormalizer.ts` +Main flow modules: -Сервисы (бизнес-логика): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- Выбор/оценка учетной записи: `open-sse/services/accountSelector.ts` -- Управление жизненным циклом контекста: `open-sse/services/contextManager.ts` -- Применение IP-фильтра: `open-sse/services/ipFilter.ts` -- Отслеживание сеанса: `open-sse/services/sessionManager.ts` -- Запросить дедупликацию: `open-sse/services/signatureCache.ts` -- Внедрение системного приглашения: `open-sse/services/systemPrompt.ts` -- Мышление управления бюджетом: `open-sse/services/thinkingBudget.ts` - — Маршрутизация по шаблонной модели: `open-sse/services/wildcardRouter.ts` -- Управление лимитом скорости: `open-sse/services/rateLimitManager.ts` -- Автоматический выключатель: `open-sse/services/circuitBreaker.ts` +Services (business logic): -Модули доменного уровня: +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- Доступность модели: `src/lib/domain/modelAvailability.ts` -- Правила/бюджеты затрат: `src/lib/domain/costRules.ts` -- Резервная политика: `src/lib/domain/fallbackPolicy.ts` - — Комбинированный преобразователь: `src/lib/domain/comboResolver.ts` -- Политика блокировки: `src/lib/domain/lockoutPolicy.ts` -- Механизм политики: `src/domain/policyEngine.ts` — централизованная блокировка → бюджет → резервная оценка. -- Каталог кодов ошибок: `src/lib/domain/errorCodes.ts` -- Идентификатор запроса: `src/lib/domain/requestId.ts` -- Таймаут получения: `src/lib/domain/fetchTimeout.ts` -- Запрос телеметрии: `src/lib/domain/requestTelemetry.ts` -- Соответствие/аудит: `src/lib/domain/compliance/index.ts` +Domain layer modules: + +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` - Eval runner: `src/lib/domain/evalRunner.ts` -- Сохранение состояния домена: `src/lib/db/domainState.ts` — SQLite CRUD для резервных цепочек, бюджетов, истории затрат, состояния блокировки, автоматических выключателей. +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -Модули провайдера OAuth (12 отдельных файлов в `src/lib/oauth/providers/`): +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -- Индекс реестра: `src/lib/oauth/providers/index.ts` -- Индивидуальные провайдеры: claude.ts, codex.ts, Gemini.ts, antigravity.ts, qoder.ts, qwen.ts, kimi-coding.ts, github.ts, kiro.ts, cursor.ts, kilocode.ts, cline.ts. - — Тонкая оболочка: `src/lib/oauth/providers.ts` — реэкспорт из отдельных модулей.## 3) Persistence Layer +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -База данных первичного состояния (SQLite): +## 3) Persistence Layer -- Базовая инфраструктура: `src/lib/db/core.ts` (better-sqlite3, миграции, WAL) -- Фасад реэкспорта: `src/lib/localDb.ts` (тонкий уровень совместимости для вызывающих абонентов) -- файл: `${DATA_DIR}/storage.sqlite` (или `$XDG_CONFIG_HOME/omniroute/storage.sqlite`, если установлен, иначе `~/.omniroute/storage.sqlite`) -- сущности (таблицы + пространства имен KV):ProviderConnections,ProviderNodes,modelAliases,combos,apiKeys,settings,pricing,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +Primary state DB (SQLite): -Постоянство использования: +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -- фасад: `src/lib/usageDb.ts` (модули разложены в `src/lib/usage/*`) -- Таблицы SQLite в `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` -- дополнительные артефакты файлов остаются для совместимости/отладки (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- устаревшие файлы JSON переносятся в SQLite при запуске миграции, если они присутствуют. +Usage persistence: -БД состояний домена (SQLite): +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- `src/lib/db/domainState.ts` — операции CRUD для состояния домена -- Таблицы (созданные в `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- Шаблон кэша со сквозной записью: карты в памяти являются авторитетными во время выполнения; мутации записываются синхронно в SQLite; состояние восстанавливается из БД при холодном запуске## 4) Auth + Security Surfaces +Domain State DB (SQLite): -- Аутентификация файлов cookie информационной панели: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- Генерация/проверка ключей API: `src/shared/utils/apiKey.ts` -- Секреты провайдера сохранялись в записях `providerConnections`. -- Поддержка исходящего прокси через `open-sse/utils/proxyFetch.ts` (env vars) и `open-sse/utils/networkProxy.ts` (настраивается для каждого провайдера или глобально)## 5) Cloud Sync +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start -- Инициализация планировщика: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Периодическая задача: `src/shared/services/cloudSyncScheduler.ts` -- Периодическая задача: `src/shared/services/modelSyncScheduler.ts` -- Маршрут управления: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Решения об откате принимаются `open-sse/services/accountFallback.ts` с использованием кодов состояния и эвристики сообщений об ошибках. Комбинированная маршрутизация добавляет одну дополнительную защиту: 400 на уровне поставщика, такие как сбои восходящего блока контента и проверки роли, рассматриваются как локальные сбои модели, поэтому более поздние комбинированные цели все еще могут выполняться.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -Обновление во время живого трафика выполняется внутри `open-sse/handlers/chatCore.ts` через исполнителя `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -Периодическая синхронизация запускается CloudSyncScheduler, когда облако включено.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -Файлы физического хранилища: +Physical storage files: -- основная база данных времени выполнения: `${DATA_DIR}/storage.sqlite` - — строки журнала запроса: `${DATA_DIR}/log.txt` (артефакт совместимости/отладки) -- структурированные архивы полезной нагрузки вызовов: `${DATA_DIR}/call_logs/` -- дополнительные сеансы отладки транслятора/запроса: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: API совместимости. -- `src/app/api/v1/providers/[provider]/*`: выделенные маршруты для каждого поставщика (чат, встраивания, изображения) -- `src/app/api/providers*`: CRUD поставщика, проверка, тестирование. -- `src/app/api/provider-nodes*`: управление настраиваемыми совместимыми узлами. -- `src/app/api/provider-models`: управление настраиваемыми моделями (CRUD). -- `src/app/api/models/route.ts`: API каталога моделей (псевдонимы + пользовательские модели) -- `src/app/api/oauth/*`: потоки кода OAuth/устройства. -- `src/app/api/keys*`: жизненный цикл локального ключа API. -- `src/app/api/models/alias`: управление псевдонимами. -- `src/app/api/combos*`: управление резервными комбинациями. -- `src/app/api/pricing`: переопределение цен для расчета затрат. -- `src/app/api/settings/proxy`: конфигурация прокси (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: проверка исходящего прокси-соединения (POST) -- `src/app/api/usage/*`: API использования и журналов. -- `src/app/api/sync/*` + `src/app/api/cloud/*`: облачная синхронизация и помощники для работы с облаком. -- `src/app/api/cli-tools/*`: локальные средства записи/проверки конфигурации CLI. -- `src/app/api/settings/ip-filter`: список разрешенных/блокированных IP-адресов (GET/PUT) -- `src/app/api/settings/thinking-budget`: конфигурация бюджета токена мышления (GET/PUT) -- `src/app/api/settings/system-prompt`: глобальное системное приглашение (GET/PUT) -- `src/app/api/sessions`: список активных сеансов (GET) -- `src/app/api/rate-limits`: статус ограничения скорости для каждой учетной записи (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: анализ запроса, обработка комбо, цикл выбора учетной записи. -- `open-sse/handlers/chatCore.ts`: перевод, отправка исполнителя, обработка повтора/обновления, настройка потока. -- `open-sse/executors/*`: поведение сети и формата в зависимости от поставщика.### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: реестр и оркестровка переводчика. -- Запросить переводчиков: `open-sse/translator/request/*` -- Переводчики ответов: `open-sse/translator/response/*` -- Константы формата: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: постоянная конфигурация/состояние и постоянство домена в SQLite. -- `src/lib/localDb.ts`: реэкспорт совместимости для модулей БД. -- `src/lib/usageDb.ts`: фасад истории использования/журналов вызовов поверх таблиц SQLite.## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -У каждого провайдера есть специализированный исполнитель, расширяющий BaseExecutor (в open-sse/executors/base.ts), который обеспечивает создание URL-адресов, построение заголовка, повторную попытку с экспоненциальной отсрочкой, перехватчики обновления учетных данных и метод оркестрации execute(). +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Исполнитель | Поставщик(и) | Специальная обработка | -| -------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------------------- | -| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Динамическая конфигурация URL/заголовка для каждого провайдера | -| `Антигравитационный Исполнитель` | Google Антигравитация | Пользовательские идентификаторы проекта/сеанса, повторная попытка после анализа | -| `CodexExecutor` | Кодекс OpenAI | Вводит системные инструкции, заставляет мыслить | -| `КурсорЭкзекутор` | Курсор IDE | Протокол ConnectRPC, кодировка Protobuf, подпись запроса через контрольную сумму | -| `GithubExecutor` | Второй пилот GitHub | Обновление токена Copilot, заголовки, имитирующие VSCode | -| `КироЭкзекутор` | AWS CodeWhisperer/Киро | Бинарный формат AWS EventStream → Преобразование SSE | -| `GeminiCLIExecutor` | Близнецы CLI | Цикл обновления токена Google OAuth | +### Persistence -Все остальные поставщики (включая пользовательские совместимые узлы) используют DefaultExecutor.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Провайдер | Формат | Авторизация | Поток | Непоток | Обновление токена | API использования | -| ------------------- | -------------- | ---------------------------------- | ----------------- | ------- | ----------------- | ---------------------------- | ------------------------------ | -| Клод | Клод | Ключ API / OAuth | ✅ | ✅ | ✅ | ⚠️ Только администратор | -| Близнецы | близнецы | Ключ API / OAuth | ✅ | ✅ | ✅ | ⚠️ Облачная консоль | -| Близнецы CLI | Близнецы-кли | ОАутент | ✅ | ✅ | ✅ | ⚠️ Облачная консоль | -| Антигравитация | антигравитация | ОАутент | ✅ | ✅ | ✅ | ✅ API с полной квотой | -| ОпенАИ | опенай | API-ключ | ✅ | ✅ | ❌ | ❌ | -| Кодекс | openai-ответы | ОАутент | ✅ принудительный | ❌ | ✅ | ✅ Ограничения ставок | -| Второй пилот GitHub | опенай | OAuth + токен второго пилота | ✅ | ✅ | ✅ | ✅ Снимки квот | -| Курсор | курсор | Пользовательская контрольная сумма | ✅ | ✅ | ❌ | ❌ | -| Киро | Киро | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Ограничения использования | -| Квен | опенай | ОАутент | ✅ | ✅ | ✅ | ⚠️ По запросу | -| Кодер | опенай | OAuth (базовый) | ✅ | ✅ | ✅ | ⚠️ По запросу | -| OpenRouter | опенай | API-ключ | ✅ | ✅ | ❌ | ❌ | -| ГЛМ/Кими/МиниМакс | Клод | API-ключ | ✅ | ✅ | ❌ | ❌ | -| ДипСик | опенай | API-ключ | ✅ | ✅ | ❌ | ❌ | -| Грок | опенай | API-ключ | ✅ | ✅ | ❌ | ❌ | -| xAI (Грок) | опенай | API-ключ | ✅ | ✅ | ❌ | ❌ | -| Мистраль | опенай | API-ключ | ✅ | ✅ | ❌ | ❌ | -| Растерянность | опенай | API-ключ | ✅ | ✅ | ❌ | ❌ | -| Вместе ИИ | опенай | API-ключ | ✅ | ✅ | ❌ | ❌ | -| Фейерверк ИИ | опенай | API-ключ | ✅ | ✅ | ❌ | ❌ | -| Церебра | опенай | API-ключ | ✅ | ✅ | ❌ | ❌ | -| Согласовано | опенай | API-ключ | ✅ | ✅ | ❌ | ❌ | -| NVIDIA НИМ | опенай | API-ключ | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Обнаруженные исходные форматы включают: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. -- `опенай` -- `openai-ответы` -- `Клод` -- `близнецы` +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | -Целевые форматы включают: +All other providers (including custom compatible nodes) use the `DefaultExecutor`. -- Чат OpenAI/Ответы -- Клод -- Оболочка Gemini/Gemini-CLI/Антигравитация -- Киро -- Курсор +## Provider Compatibility Matrix -В переводах используется**OpenAI в качестве хаб-формата**— все преобразования проходят через OpenAI в качестве промежуточного звена:``` +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: + +- `openai` +- `openai-responses` +- `claude` +- `gemini` + +Target formats include: + +- OpenAI chat/Responses +- Claude +- Gemini/Gemini-CLI/Antigravity envelope +- Kiro +- Cursor + +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Переводы выбираются динамически на основе формы исходной полезной нагрузки и целевого формата поставщика. +Additional processing layers in the translation pipeline: -Дополнительные уровни обработки в конвейере перевода: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Обеззараживание ответов** — удаляет нестандартные поля из ответов формата OpenAI (как потоковых, так и непотоковых) для обеспечения строгого соответствия SDK. --**Нормализация ролей**— преобразует `разработчик` → `система` для целей, отличных от OpenAI; объединяет `system` → `user` для моделей, которые отвергают системную роль (GLM, ERNIE) --**Извлечение тегов Think**— анализирует блоки `...` из содержимого в поле `reasoning_content`. --**Структурированный вывод** — преобразует OpenAI `response_format.json_schema` в `responseMimeType` + `responseSchema` Gemini.## Supported API Endpoints +## Supported API Endpoints -| Конечная точка | Формат | Обработчик | -| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------ | -| `POST /v1/chat/completions` | Чат OpenAI | `src/sse/handlers/chat.ts` | -| `POST /v1/messages` | Клод Сообщения | Тот же обработчик (определяется автоматически) | -| `POST /v1/ответы` | Ответы OpenAI | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/embeddings` | Вложения OpenAI | `open-sse/handlers/embeddings.ts` | -| `GET /v1/embeddings` | Список моделей | API-маршрут | -| `POST /v1/images/generations` | Изображения OpenAI | `open-sse/handlers/imageGeneration.ts` | -| `GET /v1/images/generations` | Список моделей | API-маршрут | -| `POST /v1/providers/{provider}/chat/completions` | Чат OpenAI | Выделенный для каждого поставщика с проверкой модели | -| `POST /v1/providers/{provider}/embeddings` | Вложения OpenAI | Выделенный для каждого поставщика с проверкой модели | -| `POST /v1/providers/{provider}/images/generations` | Изображения OpenAI | Выделенный для каждого поставщика с проверкой модели | -| `POST /v1/messages/count_tokens` | Количество жетонов Клода | API-маршрут | -| `GET /v1/models` | Список моделей OpenAI | Маршрут API (чат + встраивание + изображение + пользовательские модели) | -| `GET /api/models/catalog` | Каталог | Все модели сгруппированы по поставщику + типу | -| `POST /v1beta/models/*:streamGenerateContent` | Уроженец Близнецов | API-маршрут | -| `GET/PUT/DELETE /api/settings/proxy` | Конфигурация прокси | Конфигурация сетевого прокси | -| `POST /api/settings/proxy/test` | Подключение через прокси | Конечная точка проверки работоспособности/подключения прокси-сервера | -| `GET/POST/DELETE /api/provider-models` | Модели поставщиков | Метаданные модели поставщика, поддерживающие пользовательские и управляемые доступные модели |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -Обработчик обхода (`open-sse/utils/bypassHandler.ts`) перехватывает известные «одноразовые» запросы от Claude CLI — пинги прогрева, извлечение заголовков и подсчет токенов — и возвращает**поддельный ответ**без использования токенов вышестоящего поставщика. Это срабатывает только тогда, когда «User-Agent» содержит «claude-cli».## Request Logger Pipeline +## Bypass Handler -Регистратор запросов (`open-sse/utils/requestLogger.ts`) обеспечивает 7-этапный конвейер журналирования отладки, отключенный по умолчанию и включенный через `ENABLE_REQUEST_LOGS=true`:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Файлы записываются в `/logs//` для каждого сеанса запроса.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- Время восстановления учетной записи провайдера при ошибках переходного процесса/скорости/авторизации -- резервный аккаунт перед неудачным запросом -- откат комбинированной модели, когда текущий путь модели/провайдера исчерпан.## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- предварительная проверка и обновление с повтором для обновляемых поставщиков -- Повторная попытка 401/403 после попытки обновления по основному пути.## 3) Stream Safety +## 2) Token Expiry -- контроллер потока с поддержкой отключения -- поток перевода со сбросом в конце потока и обработкой `[DONE]` -- запасной вариант оценки использования, когда метаданные об использовании поставщика отсутствуют.## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- Обнаруживаются ошибки синхронизации, но локальное выполнение продолжается. -- планировщик имеет логику с возможностью повторных попыток, но периодическое выполнение в настоящее время по умолчанию вызывает синхронизацию с одной попыткой.## 5) Data Integrity +## 3) Stream Safety -- Миграция схемы SQLite и автоматическое обновление при запуске. -- устаревший JSON → путь совместимости миграции SQLite## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Источники видимости во время выполнения: +## 4) Cloud Sync Degradation -- журналы консоли из `src/sse/utils/logger.ts` -- агрегаты использования для каждого запроса в SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- четырехэтапный подробный захват полезной нагрузки в SQLite (`request_detail_logs`), когда `settings.detailed_logs_enabled=true` -- текстовый журнал статуса запроса в `log.txt` (необязательно/совместимо) -- дополнительные журналы глубоких запросов/трансляций в `logs/`, когда `ENABLE_REQUEST_LOGS=true` -- конечные точки использования информационной панели (`/api/usage/*`) для использования пользовательского интерфейса. +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -Подробный захват полезных данных запроса сохраняет до четырех этапов полезных данных JSON для каждого маршрутизируемого вызова: +## 5) Data Integrity -- необработанный запрос, полученный от клиента -- переведенный запрос фактически отправлен вверх по течению -- ответ провайдера реконструирован в формате JSON; потоковые ответы сжимаются до итоговой сводки плюс метаданные потока. -- окончательный ответ клиента, возвращаемый OmniRoute; потоковые ответы хранятся в той же компактной сводной форме## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- Секрет JWT (`JWT_SECRET`) обеспечивает проверку/подпись файлов cookie сеанса информационной панели. -- Начальная загрузка пароля (INITIAL_PASSWORD) должна быть явно настроена для инициализации при первом запуске. -- Секрет HMAC ключа API (`API_KEY_SECRET`) защищает сгенерированный локальный формат ключа API. -- Секреты поставщика (ключи/токены API) сохраняются в локальной базе данных и должны быть защищены на уровне файловой системы. -- Конечные точки облачной синхронизации полагаются на аутентификацию по ключу API + семантику идентификатора машины.## Environment and Runtime Matrix +## Observability and Operational Signals -Переменные среды, активно используемые кодом: +Runtime visibility sources: -- Приложение/аутентификация: `JWT_SECRET`, `INITIAL_PASSWORD` -- Хранилище: `DATA_DIR` -- Совместимое поведение узла: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Дополнительное переопределение базы хранения (Linux/macOS, если `DATA_DIR` не установлен): `XDG_CONFIG_HOME` -- Хеширование безопасности: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Ведение журнала: `ENABLE_REQUEST_LOGS` - – URL-адрес синхронизации/облака: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- Исходящий прокси: HTTP_PROXY, HTTPS_PROXY, ALL_PROXY, NO_PROXY и варианты в нижнем регистре. -- Флаги функций SOCKS5: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- Помощники платформы/среды выполнения (не специфичные для приложения конфигурации): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` и `localDb` используют одну и ту же политику базового каталога (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` → `~/.omniroute`) с миграцией устаревших файлов. -2. `/api/v1/route.ts` делегирует тот же единый конструктор каталогов, который используется `/api/v1/models` (`src/app/api/v1/models/catalog.ts`), чтобы избежать семантического дрейфа. -3. Регистратор запросов записывает полные заголовки/тело, если включен; рассматривать каталог журналов как конфиденциальный. -4. Поведение облака зависит от правильного `NEXT_PUBLIC_BASE_URL` и доступности конечной точки облака. -5. Каталог `open-sse/` публикуется как `@omniroute/open-sse`**пакет рабочего пространства npm**. Исходный код импортирует его через `@omniroute/open-sse/...` (разрешается с помощью Next.js `transpilePackages`). Пути к файлам в этом документе по-прежнему используют имя каталога open-sse/ для обеспечения единообразия. -6. В диаграммах на панели мониторинга используются**Recharts**(на основе SVG) для доступных интерактивных аналитических визуализаций (столбчатые диаграммы использования модели, таблицы разбивки поставщиков с показателями успешности). -7. В тестах E2E используется**Playwright**(`tests/e2e/`), который запускается через `npm run test:e2e`. Модульные тесты используют**Node.js Test Runner**(`tests/unit/`), запускаемый через `npm run test:unit`. Исходный код в `src/` —**TypeScript**(`.ts`/`.tsx`); рабочее пространство `open-sse/` остается JavaScript (`.js`). -8. Страница настроек разделена на 5 вкладок: Безопасность, Маршрутизация (6 глобальных стратегий: сначала заполнение, циклический анализ, p2c, случайная, наименее используемая, оптимизация затрат), Устойчивость (редактируемые ограничения скорости, автоматический выключатель, политики), AI (продумывание бюджета, системные подсказки, кеш подсказок), Дополнительно (прокси).## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- Сборка из исходного кода: `npm run build` - — Создайте образ Docker: `docker build -t omniroute .` -- Запустите службу и проверьте: +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: - `GET /api/settings` - `GET /api/v1/models` -- Целевой базовый URL-адрес CLI должен быть `http://:20128/v1`, если `PORT=20128`. +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/ru/docs/FEATURES.md b/docs/i18n/ru/docs/FEATURES.md index 5c420cc853..2876e874e1 100644 --- a/docs/i18n/ru/docs/FEATURES.md +++ b/docs/i18n/ru/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Визуальное руководство по каждому разделу панели управления OmniRoute.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Управляйте соединениями с поставщиками ИИ: поставщиками OAuth (Claude Code, Codex, Gemini CLI), поставщиками ключей API (Groq, DeepSeek, OpenRouter) и бесплатными поставщиками (Qoder, Qwen, Kiro). Учетные записи Kiro включают отслеживание кредитного баланса — оставшиеся кредиты, общий лимит и дату продления, которые отображаются в «Панель управления» → «Использование».![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Создавайте комбинации маршрутизации моделей с 6 стратегиями: приоритетной, взвешенной, циклической, случайной, наименее используемой и экономически оптимизированной. Каждая комбинация объединяет несколько моделей с автоматическим возвратом к резервной модели и включает в себя быстрые шаблоны и проверки готовности.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Комплексная аналитика использования с использованием токенов, оценками затрат, тепловыми картами активности, еженедельными диаграммами распределения и разбивкой по каждому провайдеру.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Мониторинг в режиме реального времени: время безотказной работы, память, версия, процентили задержки (p50/p95/p99), статистика кэша и состояния автоматического выключателя поставщика.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Четыре режима отладки переводов API:**Игровая площадка**(конвертер форматов),**Тестер чата**(живые запросы),**Тестовый стенд**(пакетные тесты) и**Живой монитор**(поток в реальном времени).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Тестируйте любую модель прямо с приборной панели. Выбирайте поставщика, модель и конечную точку, записывайте запросы с помощью редактора Monaco, транслируйте ответы в режиме реального времени, прерывайте поток в середине потока и просматривайте временные показатели.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Настраиваемые цветовые темы для всей панели управления. Выберите один из 7 предустановленных цветов (коралловый, синий, красный, зеленый, фиолетовый, оранжевый, голубой) или создайте собственную тему, выбрав любой шестнадцатеричный цвет. Поддерживает светлый, темный и системный режим.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Комплексная панель настроек с вкладками: +Comprehensive settings panel with tabs: --**Общие**— Системное хранилище, управление резервным копированием (экспорт/импорт базы данных). -**Внешний вид**— выбор темы (темная/светлая/системная), предустановки цветовых тем и пользовательские цвета, видимость журнала работоспособности, элементы управления видимостью элементов боковой панели. -**Безопасность** — защита конечных точек API, блокировка настраиваемых провайдеров, фильтрация IP-адресов, информация о сеансе. -**Маршрутизация**— псевдонимы моделей, ухудшение фоновых задач. -**Устойчивость**— сохранение ограничения скорости, настройка автоматического выключателя, автоматическое отключение заблокированных учетных записей, мониторинг истечения срока действия провайдера. -**Дополнительно**— переопределение конфигурации, контрольный журнал конфигурации, резервный режим деградации.![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Конфигурация инструментов искусственного кодирования в один клик: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor и Factory Droid. Функции автоматического применения/сброса конфигурации, профилей подключения и сопоставления моделей.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Панель мониторинга для обнаружения и управления агентами CLI. Показывает сетку из 14 встроенных агентов (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) с: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Статус установки**— Установлен/Не найден с определением версии. -**Значки протоколов**— stdio, HTTP и т. д. -**Пользовательские агенты**— зарегистрируйте любой инструмент CLI через форму (имя, двоичный файл, команда версии, аргументы создания). -**Сопоставление отпечатков CLI**— переключатель для каждого провайдера позволяет сопоставлять собственные подписи запросов CLI, снижая риск бана при сохранении IP-адреса прокси-сервера.--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Создавайте изображения, видео и музыку с панели управления. Поддерживает OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open и MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Регистрация запросов в режиме реального времени с фильтрацией по поставщику, модели, учетной записи и ключу API. Показывает коды состояния, использование токена, задержку и сведения об ответе.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Ваша унифицированная конечная точка API с разбивкой возможностей: завершение чата, API ответов, встраивание, генерация изображений, изменение рейтинга, транскрипция аудио, преобразование текста в речь, модерация и зарегистрированные ключи API. Интеграция Cloudflare Quick Tunnel и поддержка облачных прокси для удаленного доступа.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -Создание, определение области действия и отзыв ключей API. Каждый ключ может быть ограничен определенными моделями/поставщиками с полным доступом или разрешениями только на чтение. Визуальное управление ключами с отслеживанием использования.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Отслеживание административных действий с фильтрацией по типу действия, исполнителю, цели, IP-адресу и временной метке. Полная история событий безопасности.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Нативное настольное приложение Electron для Windows, macOS и Linux. Запускайте OmniRoute как отдельное приложение с интеграцией на панели задач, автономной поддержкой, автоматическим обновлением и установкой в ​​один клик. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Ключевые особенности: +Key features: -- Опрос готовности сервера (нет пустого экрана при холодном запуске) -- Системный трей с управлением портами -- Политика безопасности контента -- Единый замок -- Автообновление при перезагрузке -- Пользовательский интерфейс, зависящий от платформы (светофоры macOS, заголовок по умолчанию для Windows/Linux) -- Усиленная упаковка сборки Electron — символические ссылки `node_modules` в автономном пакете обнаруживаются и отклоняются перед упаковкой, что предотвращает зависимость времени выполнения от машины сборки (v2.5.5+). +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Полную документацию смотрите в [`electron/README.md`](../electron/README.md). +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/ru/docs/TROUBLESHOOTING.md b/docs/i18n/ru/docs/TROUBLESHOOTING.md index 7c9bd6b373..eeb4ac8263 100644 --- a/docs/i18n/ru/docs/TROUBLESHOOTING.md +++ b/docs/i18n/ru/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -Распространенные проблемы и решения OmniRoute.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Проблема | Решение | -| -------------------------------------------- | ---------------------------------------------------------------------------------------------- | --- | -| Первый вход в систему не работает | Установите `INITIAL_PASSWORD` в `.env` (без жестко запрограммированного значения по умолчанию) | -| Панель управления открывается не на тот порт | Установите `PORT=20128` и `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Нет журналов запросов в `logs/` | Установите `ENABLE_REQUEST_LOGS=true` | -| EACCES: в разрешении отказано | Установите `DATA_DIR=/path/to/writable/dir`, чтобы переопределить `~/.omniroute` | -| Стратегия маршрутизации не сохраняется | Обновление до версии 1.4.11+ (исправление схемы Zod для сохранения настроек) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Причина:**квота поставщика исчерпана. +**Cause:** Provider quota exhausted. -**Исправлено:** +**Fix:** -1. Проверьте трекер квот на панели управления. -2. Используйте комбо с запасными уровнями -3. Перейдите на более дешевый/бесплатный уровень.### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Причина:**квота подписки исчерпана. +### Rate Limiting -**Исправлено:** +**Cause:** Subscription quota exhausted. -- Добавлен запасной вариант: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Используйте GLM/MiniMax в качестве дешевой резервной копии.### OAuth Token Expired +**Fix:** -OmniRoute автоматически обновляет токены. Если проблемы сохраняются: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Панель управления → Провайдер → Переподключиться. -2. Удалить и заново добавить подключение провайдера--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Убедитесь, что `BASE_URL` указывает на ваш работающий экземпляр (например, `http://localhost:20128`). -2. Убедитесь, что CLOUD_URL указывает на конечную точку вашего облака (например, https://omniroute.dev). -3. Сохраняйте значения `NEXT_PUBLIC_*` в соответствии со значениями на стороне сервера.### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Симптом:**`Неожиданный токен 'd'...` на конечной точке облака для вызовов без потоковой передачи. +### Cloud `stream=false` Returns 500 -**Причина:**Восходящий поток возвращает полезные данные SSE, хотя клиент ожидает JSON. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Временное решение:**используйте `stream=true` для прямых вызовов из облака. Локальная среда выполнения включает резервный вариант SSE→JSON.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Создайте новый ключ на локальной панели управления (`/api/keys`). -2. Запустите облачную синхронизацию: Включить «Облако» → «Синхронизировать сейчас». -3. Старые/несинхронизированные ключи по-прежнему могут возвращать «401» в облаке.--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Проверьте поля времени выполнения: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. Для портативного режима: используйте целевой образ `runner-cli` (входящие в комплект CLI). -3. Для режима монтирования хоста: установите `CLI_EXTRA_PATHS` и смонтируйте каталог bin хоста как доступный только для чтения. -4. Если `installed=true` и `runnable=false`: двоичный файл был найден, но проверка работоспособности не удалась.### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Проверьте статистику использования в Личном кабинете → Использование. -2. Переключите основную модель на GLM/MiniMax. -3. Используйте уровень бесплатного пользования (Gemini CLI, Qoder) для некритических задач. -4. Установите бюджеты затрат для каждого ключа API: Панель управления → Ключи API → Бюджет.--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Установите `ENABLE_REQUEST_LOGS=true` в вашем файле `.env`. Журналы отображаются в каталоге logs/.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Основное состояние: `${DATA_DIR}/storage.sqlite` (поставщики, комбинации, псевдонимы, ключи, настройки) -- Использование: таблицы SQLite в `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + необязательные `${DATA_DIR}/log.txt` и `${DATA_DIR}/call_logs/` -- Запрос журналов: `/logs/...` (когда `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Когда автоматический выключатель провайдера разомкнут, запросы блокируются до истечения времени восстановления. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Исправлено:** +**Fix:** -1. Перейдите в**Панель управления → Настройки → Устойчивость**. -2. Проверьте карту автоматического выключателя соответствующего поставщика. -3. Нажмите**Сбросить все**, чтобы очистить все выключатели, или подождите, пока истечет время восстановления. -4. Перед сбросом убедитесь, что поставщик действительно доступен.### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Если провайдер неоднократно переходит в состояние OPEN: +### Provider keeps tripping the circuit breaker -1. Проверьте**Панель управления → Состояние → Состояние поставщика**, чтобы узнать о шаблоне сбоя. -2. Перейдите в**Настройки → Устойчивость → Профили поставщиков**и увеличьте порог отказа. -3. Проверьте, не изменил ли провайдер лимиты API или требует повторной аутентификации. -4. Проверьте телеметрию задержки — высокая задержка может привести к сбоям из-за тайм-аута.--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Убедитесь, что вы используете правильный префикс: «deepgram/nova-3» или «assemblyai/best». - – Убедитесь, что провайдер подключен в**Панель управления → Провайдеры**.### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Проверьте поддерживаемые аудиоформаты: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` - – Убедитесь, что размер файла находится в пределах ограничений поставщика (обычно < 25 МБ). -- Проверьте действительность ключа API провайдера в карточке провайдера.--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Используйте**Панель управления → Переводчик**для устранения проблем с переводом формата: +Use **Dashboard → Translator** to debug format translation issues: -| Режим | Когда использовать | -| ----------------------- | ------------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Детская площадка** | Сравните форматы ввода/вывода параллельно — вставьте ошибочный запрос, чтобы посмотреть, как он преобразуется | -| **Тестер чата** | Отправляйте живые сообщения и проверяйте всю полезную нагрузку запроса/ответа, включая заголовки | -| **Испытательный стенд** | Запустите пакетное тестирование комбинаций форматов, чтобы определить, какие переводы повреждены | -| **Живой монитор** | Наблюдайте за потоком запросов в режиме реального времени, чтобы выявить периодические проблемы с переводом | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Теги «Мышление» не отображаются**— проверьте, поддерживает ли целевой поставщик мышление и настройку бюджета на мышление. -**Отказ от вызовов инструментов**— Некоторые преобразования форматов могут удалять неподдерживаемые поля; проверить в режиме игровой площадки -**Отсутствует системное приглашение**— Клод и Близнецы по-разному обрабатывают системные приглашения; проверить вывод перевода -**SDK возвращает необработанную строку вместо объекта**— Исправлено в версии 1.1.0: очиститель ответов теперь удаляет нестандартные поля (`x_groq`, `usage_breakdown` и т. д.), которые вызывают сбои проверки OpenAI SDK Pydantic. -**GLM/ERNIE отклоняет роль `system`**— Исправлено в версии 1.1.0: нормализатор ролей автоматически объединяет системные сообщения с пользовательскими сообщениями для несовместимых моделей. -**роль `разработчика` не распознана**— исправлено в версии 1.1.0: автоматически преобразуется в `систему` для поставщиков, не поддерживающих OpenAI. -**`json_schema` не работает с Gemini**— Исправлено в версии 1.1.0: `response_format` теперь преобразуется в `responseMimeType` + `responseSchema` Gemini.--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- Автоматическое ограничение скорости применяется только к поставщикам ключей API (не OAuth/подписка). - – Убедитесь, что в разделе**Настройки → Устойчивость → Профили поставщиков**включено автоматическое ограничение скорости. - – Проверьте, возвращает ли поставщик коды состояния «429» или заголовки «Retry-After».### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -Профили провайдеров поддерживают следующие настройки: +### Tuning exponential backoff --**Базовая задержка**— Начальное время ожидания после первого сбоя (по умолчанию: 1 с). -**Макс. задержка**— максимальное время ожидания (по умолчанию: 30 с). -**Множитель**— насколько увеличить задержку за каждый последовательный сбой (по умолчанию: 2x)### Anti-thundering herd +Provider profiles support these settings: -Когда множество одновременных запросов попадают к поставщику с ограниченной скоростью, OmniRoute использует мьютекс + автоматическое ограничение скорости для сериализации запросов и предотвращения каскадных сбоев. Это происходит автоматически для поставщиков ключей API.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Некоторые пользователи OmniRoute размещают шлюз перед стеками RAG или агентов. В таких настройках часто можно увидеть странную картину: OmniRoute выглядит исправным (провайдеры работают, профили маршрутизации в порядке, предупреждений об ограничении скорости нет), но окончательный ответ все равно неправильный. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -На практике эти инциденты обычно происходят из нисходящего конвейера RAG, а не из самого шлюза. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Если вам нужен общий словарь для описания этих сбоев, вы можете использовать WFGY IssueMap, внешний текстовый ресурс лицензии MIT, который определяет шестнадцать повторяющихся шаблонов сбоев RAG/LLM. На высоком уровне это охватывает: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- дрейф поиска и нарушение границ контекста -- пустые или устаревшие индексы и векторные хранилища -- встраивание против семантического несоответствия -- быстрая сборка и проблемы с контекстными окнами -- коллапс логики и самоуверенные ответы -- длинная цепочка и сбои координации агентов -- память нескольких агентов и дрейф ролей -- проблемы с развертыванием и загрузочным заказом +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -Идея проста: +The idea is simple: -1. Когда вы расследуете плохой ответ, зафиксируйте: - - задача и запрос пользователя - - комбинация маршрутов или провайдеров в OmniRoute - - любой контекст RAG, используемый в дальнейшем (полученные документы, вызовы инструментов и т. д.) -2. Сопоставьте инцидент с одним или двумя номерами WFGY IssueMap («№1»… «№16»). -3. Сохраните номер на своей информационной панели, в книге задач или в системе отслеживания инцидентов рядом с журналами OmniRoute. -4. Используйте соответствующую страницу WFGY, чтобы решить, нужно ли вам изменить стек RAG, средство извлечения или стратегию маршрутизации. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -Полный текст и конкретные рецепты находятся здесь (лицензия MIT, только текст): +Full text and concrete recipes live here (MIT license, text only): -[README WFGY IssueMap](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) +[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -Вы можете игнорировать этот раздел, если вы не используете RAG или конвейеры агентов за OmniRoute.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**Проблемы с GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Архитектура**: внутренние подробности см. в [`docs/ARCHITECTURE.md`](ARCHITECTURE.md). -**Справочник API**: см. [`docs/API_REFERENCE.md`](API_REFERENCE.md) для всех конечных точек. -**Панель состояния**: проверьте**Панель управления → Здоровье**, чтобы узнать состояние системы в режиме реального времени. -**Переводчик**: используйте**Панель управления → Переводчик**для устранения проблем с форматом. +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/ru/llm.txt b/docs/i18n/ru/llm.txt new file mode 100644 index 0000000000..35e2273e73 --- /dev/null +++ b/docs/i18n/ru/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Русский) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Обзор + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Безопасность +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/sk/README.md b/docs/i18n/sk/README.md index 3770f15864..424aaecd52 100644 --- a/docs/i18n/sk/README.md +++ b/docs/i18n/sk/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Váš univerzálny proxy server API – jeden koncový bod, 60+ poskytovateľov, nulové prestoje. Teraz s**MCP Server (25 nástrojov)**,**Protokol A2A**,**Pamäť/systémy zručností**a**Electron Desktop App**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Dokončenia chatu • Vloženie • Generovanie obrázkov • Video • Hudba • Zvuk • Zmena poradia •**Vyhľadávanie na webe**• Server MCP • Protokol A2A • 100 % TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Váš univerzálny proxy server API – jeden koncový bod, 60+ poskytovateľov [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Webová stránka](https://omniroute.online) • [🚀 Rýchly štart](#-rýchly štart) • [💡 Funkcie](#-kľúčové-funkcie) • [📖 Dokumenty](#-dokumentácia) • [💰 Ceny](#-ceny-na prvý pohľad) • [🬒] WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Dostupné v:**🇺🇸 [angličtina](README.md) | 🇧🇷 [Português (Brazília)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Taliansko](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [nemecky](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Maďarčina](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugalsko)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,556 +60,629 @@ _Váš univerzálny proxy server API – jeden koncový bod, 60+ poskytovateľov ## 📸 Dashboard Preview - -Kliknutím zobrazíte snímky obrazovky hlavného panela +
+Click to see dashboard screenshots -| Strana | Snímka obrazovky | -| ---------------------- | ---------------------------------------------------- | ---------- | -| **Poskytovatelia** | ![Poskytovatelia](docs/screenshots/01-providers.png) | -| **Kombá** | ![Combos](docs/screenshots/02-combos.png) | -| **Analytika** | ![Analytics](docs/screenshots/03-analytics.png) | -| **Zdravie** | ![Zdravie](docs/screenshots/04-health.png) | -| **Prekladateľ** | ![Translator](docs/screenshots/05-translator.png) | -| **Nastavenia** | ![Nastavenia](docs/screenshots/06-settings.png) | -| **Nástroje CLI** | ![Nástroje CLI](docs/screenshots/07-cli-tools.png) | -| **Denníky používania** | ![Použitie](docs/screenshots/08-usage.png) | -| **Koncové body** | ![Koncové body](docs/screenshots/09-endpoint.png) |
| +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | + + --- ### 🤖 Free AI Provider for your favorite coding agents -_Pripojte akýkoľvek nástroj IDE alebo CLI poháňaný AI cez OmniRoute – bezplatnú bránu API pre neobmedzené kódovanie._ - - - - - -OpenClaw
-OpenClaw -

-⭐ 205 tis. - - - -NanoBot
-NanoBot -

-⭐ 20,9 000 - - - -PicoClaw
-PicoClaw -

-⭐ 14,6 000 - - - -ZeroClaw
-ZeroClaw -

-⭐ 9,9 000 - - - -Železný pazúr
-IronClaw -

-⭐ 2,1 tis. - - - - - -OpenCode
-OpenCode -

-⭐ 106 tis. - - - -Codex CLI
-Codex CLI -

-⭐ 60,8 tis. - - - -Kód Claude
-Claude Code -

-⭐ 67,3 tis. - - - -Gemini CLI
-Gemini CLI -

-⭐ 94,7 tis. - - - -Kilogramový kód
-Kilogram -

-⭐ 15,5 000 - - +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ + + + + + + + + + + + + + + +
+ + OpenClaw
+ OpenClaw +

+ ⭐ 205K +
+ + NanoBot
+ NanoBot +

+ ⭐ 20.9K +
+ + PicoClaw
+ PicoClaw +

+ ⭐ 14.6K +
+ + ZeroClaw
+ ZeroClaw +

+ ⭐ 9.9K +
+ + IronClaw
+ IronClaw +

+ ⭐ 2.1K +
+ + OpenCode
+ OpenCode +

+ ⭐ 106K +
+ + Codex CLI
+ Codex CLI +

+ ⭐ 60.8K +
+ + Claude Code
+ Claude Code +

+ ⭐ 67.3K +
+ + Gemini CLI
+ Gemini CLI +

+ ⭐ 94.7K +
+ + Kilo Code
+ Kilo Code +

+ ⭐ 15.5K +
-📡 Všetci agenti sa pripájajú cez http://localhost:20128/v1 alebo http://cloud.omniroute.online/v1 — jedna konfigurácia, neobmedzené modely a kvóta--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Prestaňte plytvať peniazmi a dosahovať limity:** +**Stop wasting money and hitting limits:** -- Kvóta odberu vyprší nevyužitá každý mesiac - – Obmedzenia sadzieb vám bránia v kódovaní uprostred - – Drahé rozhrania API (20 – 50 USD/mesiac na poskytovateľa) -- Manuálne prepínanie medzi poskytovateľmi +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute to rieši:** +**OmniRoute solves this:** -- ✅**Maximalizujte odbery**- Sledujte kvótu, pred resetovaním použite každý bit -- ✅**Automatická záloha**- Predplatné → Kľúč API → Lacné → Bezplatne, nulové prestoje -- ✅**Viacnásobný účet**- Obojstranne medzi účtami na poskytovateľa -- ✅**Universal**- Funguje s Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, akýmkoľvek nástrojom CLI--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Pripojte sa k našej komunite!**[WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Získajte pomoc, zdieľajte tipy a buďte informovaní. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Web**: [omniroute.online](https://omniroute.online) -–**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -–**Problémy**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Skupina komunity](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Prispievanie**: Pozrite si [CONTRIBUTING.md](CONTRIBUTING.md), otvorte PR alebo si vyberte „dobré prvé číslo“ -**Pôvodný projekt**: [9router by decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Pri otváraní problému spustite príkaz system-info a pripojte vygenerovaný súbor:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Toto vygeneruje súbor `system-info.txt` s vašou verziou Node.js, verziou OmniRoute, podrobnosťami operačného systému, nainštalovanými nástrojmi CLI (qoder, gemini, claude, codex, antigravity, droid atď.), stavom Docker/PM2 a systémovými balíkmi – všetko, čo potrebujeme na rýchlu reprodukciu vášho problému. Pripojte súbor priamo k vášmu problému na GitHub.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Každý vývojár, ktorý používa nástroje AI, čelí týmto problémom denne.**OmniRoute bol vytvorený tak, aby ich všetky vyriešil – od prekročenia nákladov po regionálne bloky, od prerušených tokov OAuth po operácie protokolov a pozorovateľnosť podniku. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. - -💸 1. „Platím za drahé predplatné, ale stále ma vyrušujú limity“ +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Vývojári platia za Claude Pro, Codex Pro alebo GitHub Copilot 20 – 200 dolárov mesačne. Aj pri platení má kvóta strop – 5 hodín používania, týždenné limity alebo limity za minútu. Počas relácie kódovania poskytovateľ prestane reagovať a vývojár stráca tok a produktivitu. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Ako to rieši OmniRoute:** +**How OmniRoute solves it:** --**Inteligentný 4-úrovňový záložný systém**– Ak sa vyčerpá kvóta predplatného, automaticky sa presmeruje na kľúč API → Lacné → Zadarmo s nulovým manuálnym zásahom -–**Sledovanie limitov poskytovateľa**– Snímky kvót vo vyrovnávacej pamäti sa obnovujú podľa plánu na strane servera (predvolená hodnota `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) s možnosťou manuálneho obnovenia v používateľskom rozhraní -–**Podpora viacerých účtov**– Viacero účtov na poskytovateľa s automatickým opakovaním – keď sa jeden minie, prepne sa na ďalší --**Vlastné kombá**– Prispôsobiteľné záložné reťazce s 9 stratégiami vyvažovania (prioritná, vážená, na prvom mieste, s opakovaným výberom, P2C, náhodná, najmenej používaná, nákladovo optimalizovaná, striktne náhodná) --**Codex Business Quotas**— Monitorovanie kvót pracovného priestoru pre firmy/tím priamo na paneli
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard - -🔌 2. „Potrebujem použiť viacerých poskytovateľov, ale každý má iné API“ + -OpenAI používa jeden formát, Claude (Anthropic) iný a Gemini ďalší. Ak chce vývojár testovať modely od rôznych poskytovateľov alebo medzi nimi záložné riešenie, musí prekonfigurovať súpravy SDK, zmeniť koncové body, vysporiadať sa s nekompatibilnými formátmi. Vlastní poskytovatelia (FriendLI, NIM) majú neštandardné modelové koncové body. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Ako to rieši OmniRoute:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Unified Endpoint**– Jediný `http://localhost:20128/v1` slúži ako proxy pre všetkých 60+ poskytovateľov --**Formátový preklad**— Automatický a transparentný: OpenAI ↔ Claude ↔ Gemini ↔ Responses API -–**Odstraňovanie odozvy**– Odstraňuje neštandardné polia (`x_groq`, `usage_breakdown`, `service_tier`), ktoré porušujú OpenAI SDK v1.83+ --**Normalizácia rolí**– Konvertuje „vývojár“ → „systém“ pre poskytovateľov, ktorí nie sú OpenAI; `systém` → `používateľ` pre GLM/ERNIE -–**Think Tag Extraction**– Extrahuje bloky „“ z modelov ako DeepSeek R1 do štandardizovaného „reasoning_content“ --**Štruktúrovaný výstup pre Gemini**— `json_schema` → automatická konverzia `responseMimeType`/`responseSchema` --**`stream` predvolene na `false`**— Zosúladí sa so špecifikáciou OpenAI, čím sa zabráni neočakávanému SSE v súpravách Python/Rust/Go SDK
+**How OmniRoute solves it:** - -🌐 3. „Môj poskytovateľ AI blokuje môj región/krajinu“ +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Poskytovatelia ako OpenAI/Codex blokujú prístup z určitých geografických oblastí. Počas pripojení OAuth a API sa používateľom zobrazujú chyby ako „unsupported_country_region_territory“. To je frustrujúce najmä pre vývojárov z rozvojových krajín. + -**Ako to rieši OmniRoute:** +
+🌐 3. "My AI provider blocks my region/country" --**Konfigurácia proxy servera na troch úrovniach**– Konfigurovateľný server proxy na 3 úrovniach: globálny (celá prevádzka), podľa jednotlivých poskytovateľov (iba jeden poskytovateľ) a podľa pripojenia/kľúča --**Farebné odznaky proxy**— Vizuálne indikátory: 🢢 globálny proxy, 🟡 proxy poskytovateľa, 🔵 proxy pripojenia, vždy zobrazuje IP -–**Výmena tokenov OAuth cez server proxy**– tok OAuth prechádza aj cez server proxy, čím sa rieši „unsupported_country_region_territory“ --**Testy pripojenia cez proxy**– Testy pripojenia používajú nakonfigurovaný proxy (už žiadne priame obchádzanie) --**Podpora SOCKS5**— Úplná podpora proxy SOCKS5 pre odchádzajúce smerovanie --**TLS Fingerprint Spoofing**– Odtlačok prsta TLS podobný prehliadaču cez `wreq-js` na obídenie detekcie robotov --**🔏 CLI Fingerprint Matching**– Zmení poradie hlavičiek a polí tela tak, aby sa zhodovali s natívnymi binárnymi podpismi CLI, čím sa výrazne zníži riziko označenia účtu. IP proxy servera je zachovaná – získate súčasne utajené**aj**maskovanie IP
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. - -🆓 4. „Chcem používať AI na kódovanie, ale nemám peniaze“ +**How OmniRoute solves it:** -Nie každý môže platiť 20 – 200 $ mesačne za predplatné AI. Študenti, vývojári z rozvíjajúcich sa krajín, fanúšikovia a nezávislí pracovníci potrebujú prístup ku kvalitným modelom za nulové náklady. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Ako to rieši OmniRoute:** + --**Zabudovaní poskytovatelia bezplatnej úrovne**— Natívna podpora pre 100 % bezplatných poskytovateľov: Qoder (5 neobmedzených modelov cez OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 neobmedzené modely: qco,3-qwender-lash,3-qwender-lash qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID zadarmo), Gemini CLI (180 000 tokenov mesačne zadarmo) --**Ollama Cloud**– modely Ollama hostené v cloude na `api.ollama.com` s bezplatnou úrovňou „Light use“; použite predponu `ollamacloud/` --**Len bezplatné kombá**– Reťaz `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = 0 USD/mesiac s nulovými prestojmi --**Voľný prístup NVIDIA NIM**— ~40 RPM pre vývojárov - navždy bezplatný prístup k viac ako 70 modelom na stránke build.nvidia.com (prechod z kreditov na limity čistej sadzby) --**Cost Optimized Strategy**– Stratégia smerovania, ktorá automaticky vyberie najlacnejšieho dostupného poskytovateľa +
+🆓 4. "I want to use AI for coding but I have no money" - -🔒 5. „Potrebujem chrániť svoju bránu AI pred neoprávneným prístupom“ +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -Pri vystavení brány AI do siete (LAN, VPS, Docker) môže ktokoľvek s adresou spotrebovať tokeny/kvótu vývojára. Bez ochrany sú rozhrania API náchylné na nesprávne použitie, rýchle vloženie a zneužitie. +**How OmniRoute solves it:** -**Ako to rieši OmniRoute:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider -–**Správa kľúčov API**– Generovanie, rotácia a rozsah podľa poskytovateľa pomocou vyhradenej stránky `/dashboard/api-manager` -–**Povolenia na úrovni modelu**– Obmedzte kľúče API na konkrétne modely (`openai/*`, vzory zástupných znakov) pomocou prepínača Povoliť všetko/obmedziť -–**API Endpoint Protection**– Vyžadovať kľúč pre `/v1/models` a blokovať konkrétnych poskytovateľov v zozname --**Auth Guard + ochrana CSRF**– všetky smerovanie dashboardu chránené middlewarom „withAuth“ + tokenmi CSRF --**Rate Limiter**— Obmedzenie rýchlosti na IP pomocou konfigurovateľných okien --**IP Filtering**— Zoznam povolených/blokovaných pre riadenie prístupu --**Prompt Injection Guard**– Dezinfekcia proti škodlivým vzorom výzvy --**Šifrovanie AES-256-GCM**— Prihlasovacie údaje sú v pokoji zašifrované
+ - -🛑 6. „Môj poskytovateľ zlyhal a stratil som tok kódovania“ +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -Poskytovatelia AI sa môžu stať nestabilnými, vrátiť chyby 5xx alebo dosiahnuť dočasné limity sadzieb. Ak vývojár závisí od jedného poskytovateľa, bude prerušený. Bez ističov môžu opakované pokusy zlyhať aplikáciu. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Ako to rieši OmniRoute:** +**How OmniRoute solves it:** --**Istič pre každý model**- Automatické otváranie/zatváranie s konfigurovateľnými prahmi a ochladzovaním (zatvorené/otvorené/polootvorené), s rozsahom pre každý model, aby sa predišlo kaskádovým blokom --**Exponenciálne stiahnutie**— Postupné oneskorenie opakovania --**Anti-Thundering Herd**- ochrana Mutex + semafor proti súbežným opakovaným búrkam --**Combo Fallback Chains**– Ak primárny poskytovateľ zlyhá, automaticky prepadne reťazcom bez akéhokoľvek zásahu --**Combo Circuit Breaker**– Automaticky deaktivuje zlyhávajúcich poskytovateľov v rámci kombinovaného reťazca -–**Health Dashboard**– Monitorovanie dostupnosti, stavy ističov, blokovania, štatistiky vyrovnávacej pamäte, latencia p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest - -🔧 7. „Konfigurácia každého nástroja AI je únavná a opakovaná“ + -Vývojári používajú Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Každý nástroj potrebuje inú konfiguráciu (API endpoint, kľúč, model). Prekonfigurovanie pri zmene poskytovateľa alebo modelu je strata času. +
+🛑 6. "My provider went down and I lost my coding flow" -**Ako to rieši OmniRoute:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**CLI Tools Dashboard**– Vyhradená stránka s nastavením jedným kliknutím pre Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline -–**GitHub Copilot Config Generator**– Generuje `chatLanguageModels.json` pre kód VS s hromadným výberom modelu --**Sprievodca registráciou**– Sprievodca nastavením v 4 krokoch pre začínajúcich používateľov -–**Jeden koncový bod, všetky modely**– Nakonfigurujte `http://localhost:20128/v1` raz a získajte prístup k 60+ poskytovateľom
+**How OmniRoute solves it:** - -🔑 8. „Správa tokenov OAuth od viacerých poskytovateľov je peklo“ +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot – všetky používajú OAuth 2.0 s tokenmi, ktorých platnosť sa končí. Vývojári sa musia neustále znovu overovať, riešiť problémy `client_secret missing`, `redirect_uri_mismatch` a zlyhania na vzdialených serveroch. Obzvlášť problematické je OAuth na LAN/VPS. + -**Ako to rieši OmniRoute:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Automatická obnova tokenov**– Tokeny OAuth sa pred vypršaním platnosti obnovujú na pozadí --**Vstavaný OAuth 2.0 (PKCE)**– automatický tok pre Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder -–**Multi-Auth OAuth**– Viaceré účty na poskytovateľa prostredníctvom extrakcie tokenov JWT/ID -–**Oprava OAuth LAN/Remote**– detekcia súkromnej adresy IP pre `redirect_uri` + manuálny režim adresy URL pre vzdialené servery -–**OAuth Behind Nginx**– Používa `window.location.origin` na reverznú kompatibilitu proxy -–**Príručka vzdialeného OAuth**– Podrobný sprievodca povereniami Google Cloud na VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. - -📊 9. "Neviem, koľko míňam ani kde" +**How OmniRoute solves it:** -Vývojári využívajú viacerých platených poskytovateľov, ale nemajú jednotný pohľad na výdavky. Každý poskytovateľ má svoj vlastný informačný panel fakturácie, ale neexistuje žiadne konsolidované zobrazenie. Neočakávané náklady sa môžu nahromadiť. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Ako to rieši OmniRoute:** + -–**Informačný panel analýzy nákladov**– sledovanie nákladov na token a správa rozpočtu podľa poskytovateľa --**Obmedzenia rozpočtu na úroveň**– Strop výdavkov na úroveň, ktorý spúšťa automatické záložné právo --**Konfigurácia cien za model**– Konfigurovateľné ceny za model --**Štatistiky používania na kľúč API**– Počet žiadostí a časová pečiatka posledného použitia na kľúč -–**Panel Analytics**– štatistické karty, graf používania modelu, tabuľka poskytovateľov s mierami úspešnosti a latenciou +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" - -🐛 10. „Nedokážem diagnostikovať chyby a problémy vo volaniach AI“ +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Keď hovor zlyhá, vývojár nevie, či to bol limit sadzby, vypršaný token, nesprávny formát alebo chyba poskytovateľa. Fragmentované protokoly cez rôzne terminály. Bez pozorovateľnosti je ladenie metódou pokus-omyl. +**How OmniRoute solves it:** -**Ako to rieši OmniRoute:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker -–**Panel jednotných protokolov**– 4 karty: Protokoly žiadostí, Protokoly proxy, Protokoly auditu, Konzola --**Console Log Viewer**— Prehliadač v štýle terminálu v reálnom čase s farebne odlíšenými úrovňami, automatickým posúvaním, vyhľadávaním a filtrovaním --**Proxy protokoly SQLite**— Trvalé protokoly, ktoré prežijú reštart servera --**Translator Playground**– 4 režimy ladenia: Playground (preklad formátu), Chat Tester (spiatočný), Test Bench (dávka), Live Monitor (v reálnom čase) -–**Požiadať o telemetriu**– latencia p50/p95/p99 + sledovanie X-request-Id --**Protokolovanie založené na súboroch s rotáciou**– Protokoly aplikácií rotujú podľa veľkosti, dní uchovávania a počtu archívov; Artefakty protokolu hovorov rotujú podľa dní uchovávania a počtu súborov --**Správa systémových informácií**— `npm run system-info` vygeneruje `system-info.txt` s vaším úplným prostredím (verzia uzla, verzia OmniRoute, OS, nástroje CLI, stav Docker/PM2). Pripojte ho pri nahlasovaní problémov na okamžité triedenie.
+ - -🏗️ 11. „Nasadenie a údržba brány je zložitá“ +
+📊 9. "I don't know how much I'm spending or where" -Inštalácia, konfigurácia a údržba AI proxy v rôznych prostrediach (lokálne, VPS, Docker, cloud) je náročná na prácu. Problémy ako pevne zakódované cesty, „EACCES“ v adresároch, konflikty portov a zostavy naprieč platformami zvyšujú trenie. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Ako to rieši OmniRoute:** +**How OmniRoute solves it:** --**Globálna inštalácia npm**– `npm install -g omniroute && omniroute` – hotovo --**Docker Multi-Platform**– natívne AMD64 + ARM64 (Apple Silicon, AWS Graviton, Raspberry Pi) --**Profily Docker Compose**– `base` (bez nástrojov CLI) a `cli` (s Claude Code, Codex, OpenClaw) --**Electron Desktop App**– natívna aplikácia pre Windows/macOS/Linux so systémovou lištou, automatickým spustením, offline režimom --**Split-Port Mode**– API a Dashboard na samostatných portoch pre pokročilé scenáre (reverzný proxy, kontajnerová sieť) --**Cloud Sync**— Synchronizácia konfigurácie medzi zariadeniami cez Cloudflare Workers --**DB Backups**— Automatické zálohovanie, obnovenie, export a import všetkých nastavení s `DISABLE_SQLITE_AUTO_BACKUP` pre externe spravované zálohy
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency - -🌍 12. "Rozhranie je len v angličtine a môj tím nehovorí po anglicky" + -Tímy v neanglicky hovoriacich krajinách, najmä v Latinskej Amerike, Ázii a Európe, zápasia s rozhraním iba v angličtine. Jazykové bariéry znižujú prijatie a zvyšujú chyby v konfigurácii. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Ako to rieši OmniRoute:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Dashboard i18n — 30 jazykov**— Všetkých 500+ kláves preložených vrátane arabčiny, bulharčiny, dánčiny, nemčiny, španielčiny, fínčiny, francúzštiny, hebrejčiny, hindčiny, maďarčiny, indonézštiny, taliančiny, japončiny, kórejčiny, malajčiny, holandčiny, nórčiny, poľštiny, portugalčiny (PT/BR), rumunčiny, ruštiny, slovenčiny, švédčiny, thajčiny, ukrajinčiny, vietnamčiny, angličtiny --**Podpora RTL**— Podpora sprava doľava pre arabčinu a hebrejčinu --**Viacjazyčné README**— 30 kompletných prekladov dokumentácie --**Language Selector**– ikona zemegule v hlavičke pre prepínanie v reálnom čase
+**How OmniRoute solves it:** - -🔄 13. „Potrebujem viac než len chat – potrebujem vloženie, obrázky, zvuk“ +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -AI nie je len dokončenie chatu. Vývojári potrebujú generovať obrázky, prepisovať zvuk, vytvárať vloženia pre RAG, meniť hodnotenie dokumentov a moderovať obsah. Každé API má iný koncový bod a formát. + -**Ako to rieši OmniRoute:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Vložené**— `/v1/embeddings` so 6 poskytovateľmi a 9+ modelmi --**Generácia obrázkov**— `/v1/images/generations` s 10 poskytovateľmi a 20+ modelmi (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Text-to-Video**— `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) a SD WebUI --**Text-to-Music**— `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) --**Audio Transscription**— `/v1/audio/transscriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Text-to-Speech**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + existujúci poskytovatelia --**Moderácie**— `/v1/moderations` — Kontroly bezpečnosti obsahu --**Prehodnotenie**— `/v1/rerank` — Zmena poradia podľa relevantnosti dokumentu --**Responses API**– plná podpora `/v1/responses` pre kódex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. - -🧪 14. „Nemám možnosť testovať a porovnávať kvalitu medzi modelmi“ +**How OmniRoute solves it:** -Vývojári chcú vedieť, ktorý model je pre ich prípad použitia najlepší – kód, preklad, zdôvodnenie – ale manuálne porovnávanie je pomalé. Neexistujú žiadne integrované nástroje hodnotenia. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**Ako to rieši OmniRoute:** + --**Hodnotenia LLM**– testovanie zlatej sady s 10 predinštalovanými prípadmi zahŕňajúcimi pozdravy, matematiku, geografiu, generovanie kódu, súlad s JSON, preklad, označenie, odmietnutie bezpečnosti -–**4 stratégie zhody**– „presná“, „obsahuje“, „regulárny výraz“, „vlastné“ (funkcia JS) --**Testovacia lavica pre prekladateľské ihrisko**– dávkové testovanie s viacerými vstupmi a očakávanými výstupmi, porovnanie medzi poskytovateľmi --**Chat Tester**– celý spiatočný výlet s vykresľovaním vizuálnej odozvy --**Live Monitor**– tok všetkých požiadaviek prechádzajúcich cez server proxy v reálnom čase +
+🌍 12. "The interface is English-only and my team doesn't speak English" - -📈 15. „Potrebujem škálovať bez straty výkonu“ +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -Keďže objem žiadostí rastie, bez ukladania rovnakých otázok do vyrovnávacej pamäte vznikajú duplicitné náklady. Bez idempotencie duplikát požaduje spracovanie odpadu. Musia sa dodržiavať limity sadzieb na poskytovateľa. +**How OmniRoute solves it:** -**Ako to rieši OmniRoute:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Sémantická vyrovnávacia pamäť**– Dvojvrstvová vyrovnávacia pamäť (podpis + sémantická) znižuje náklady a latenciu --**Idempotencia požiadavky**– 5-sekundové deduplikačné okno pre identické požiadavky -–**Detekcia limitu rýchlosti**– RPM, minimálna medzera a maximálne súbežné sledovanie jednotlivých poskytovateľov --**Upraviteľné limity frekvencie**– Konfigurovateľné predvolené hodnoty v Nastaveniach → Odolnosť s perzistenciou --**Cache na overenie kľúča API**— 3-vrstvová vyrovnávacia pamäť pre produkčný výkon -–**Panel zdravia s telemetriou**– latencia p50/p95/p99, štatistiky vyrovnávacej pamäte, doba prevádzky
+ - -🤖 16. „Chcem globálne ovládať správanie modelu“ +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Vývojári, ktorí chcú všetky odpovede v konkrétnom jazyku, so špecifickým tónom alebo chcú obmedziť tokeny uvažovania. Konfigurovať to v každom nástroji/požiadavke je nepraktické. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Ako to rieši OmniRoute:** +**How OmniRoute solves it:** --**System Prompt Injection**– Globálna výzva aplikovaná na všetky požiadavky -–**Thinking Budget Validation**– Zdôvodnenie riadenia prideľovania tokenov na žiadosť (priechodné, automatické, vlastné, adaptívne) --**9 stratégií smerovania**— Globálne stratégie, ktoré určujú spôsob distribúcie požiadaviek --**Wildcard Router**– vzory „poskytovateľ/*“ sa dynamicky smerujú k akémukoľvek poskytovateľovi --**Prepínač povoliť/zakázať kombo**— Prepínajte kombinácie priamo z ovládacieho panela --**Provider Toggle**— Povolenie/zakázanie všetkých pripojení pre poskytovateľa jedným kliknutím -–**Blokovaní poskytovatelia**– vylúčte konkrétnych poskytovateľov zo zoznamu `/v1/models`
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex - -🧰 17. „Potrebujem nástroje MCP ako prvotriedne možnosti produktu“ + -Mnohé brány AI odhaľujú MCP iba ako skrytý detail implementácie. Tímy potrebujú viditeľnú a spravovateľnú operačnú vrstvu. +
+🧪 14. "I have no way to test and compare quality across models" -**Ako to rieši OmniRoute:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP sa zobrazí na navigačnom paneli a na karte protokolu koncového bodu -- Vyhradená stránka správy MCP s procesmi, nástrojmi, rozsahmi a auditom -- Vstavaný rýchly štart pre `omniroute --mcp` a registráciu klienta
+**How OmniRoute solves it:** - -🧠 18. „Potrebujem orchestráciu A2A s cestami synchronizácie + streamovania“ +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Pracovné postupy agentov vyžadujú priame odpovede a dlhotrvajúce streamované vykonávanie s kontrolou životného cyklu. + -**Ako to rieši OmniRoute:** +
+📈 15. "I need to scale without losing performance" -– A2A JSON-RPC koncový bod (`POST /a2a`) s `správou/odoslaním` a `správou/streamom` -- SSE streaming so šírením koncového stavu -- API životného cyklu úloh pre `tasks/get` a `tasks/cancel`
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. - -🛰️ 19. „Potrebujem skutočné zdravie procesu MCP, nie uhádnutý stav“ +**How OmniRoute solves it:** -Operačné tímy potrebujú vedieť, či je MCP skutočne nažive, nielen to, či je dostupné API. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**Ako to rieši OmniRoute:** + -- Súbor srdcového tepu za behu s PID, časovými pečiatkami, transportom, počtom nástrojov a režimom rozsahu -- Stavové rozhranie MCP API, ktoré kombinuje srdcový tep + nedávnu aktivitu -- Stavové karty používateľského rozhrania pre sviežosť procesu / dostupnosti / tepu +
+🤖 16. "I want to control model behavior globally" - -📋 20. „Potrebujem auditovateľné spustenie nástroja MCP“ +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Keď nástroje mutujú konfiguráciu alebo spúšťajú akcie operácií, tímy potrebujú forenznú sledovateľnosť. +**How OmniRoute solves it:** -**Ako to rieši OmniRoute:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- Záznamy auditu podporované SQLite pre volania nástrojov MCP -- Filtre podľa nástroja, úspechu/neúspechu, kľúča API a stránkovania -- Tabuľka auditu palubnej dosky + štatistické koncové body pre automatizáciu
+ - -🔐 21. „Na integráciu potrebujem povolenia MCP v rozsahu“ +
+🧰 17. "I need MCP tools as first-class product capabilities" -Rôzni klienti by mali mať najmenej privilegovaný prístup ku kategóriám nástrojov. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Ako to rieši OmniRoute:** +**How OmniRoute solves it:** -- 10 zrnitých rozsahov MCP pre kontrolovaný prístup k nástrojom -- Presadzovanie rozsahu a viditeľnosť v používateľskom rozhraní správy MCP -- Bezpečná východisková poloha pre prevádzkové nástroje
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding - -⚙️ 22. „Potrebujem prevádzkové kontroly bez premiestňovania“ + -Tímy potrebujú rýchle zmeny runtime počas incidentov alebo nákladových udalostí. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**Ako to rieši OmniRoute:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Aktivácia komba prepínača priamo z ovládacieho panela MCP -- Použite profily odolnosti z preddefinovaných balíkov politík -- Resetujte stav ističa z rovnakého ovládacieho panela
+**How OmniRoute solves it:** - -🔄 23. „Potrebujem viditeľnosť a zrušenie životného cyklu úlohy A2A naživo“ +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Bez viditeľnosti životného cyklu sa incidenty úloh ťažko triedia. + -**Ako to rieši OmniRoute:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Zoznam úloh / filtrovanie podľa stavu / zručnosti so stránkovaním -- Rozbalenie metadát úloh, udalostí a artefaktov -- Koncový bod zrušenia úlohy a akcia používateľského rozhrania s potvrdením
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. - -🌊 24. „Potrebujem aktívne metriky streamu pre načítanie A2A“ +**How OmniRoute solves it:** -Streamovanie pracovných tokov vyžaduje operačný prehľad o súbežnosti a živých pripojeniach. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**Ako to rieši OmniRoute:** + -- Aktívne počítadlá toku integrované do stavu A2A -- Časová pečiatka poslednej úlohy a počet jednotlivých štátov -- Karty palubnej dosky A2A na monitorovanie operácií v reálnom čase +
+📋 20. "I need auditable MCP tool execution" - -🪪 25. „Potrebujem štandardné vyhľadávanie agentov pre klientov“ +When tools mutate config or trigger ops actions, teams need forensic traceability. -Externí klienti a orchestrátori potrebujú strojovo čitateľné metadáta na integráciu. +**How OmniRoute solves it:** -**Ako to rieši OmniRoute:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -– Karta agenta vystavená na adrese `/.well-known/agent.json` -- Schopnosti a zručnosti zobrazené v používateľskom rozhraní správy -- API stavu A2A obsahuje metaúdaje zisťovania pre automatizáciu
+ - -🧭 26. „Potrebujem viditeľnosť protokolu v používateľskom rozhraní produktu“ +
+🔐 21. "I need scoped MCP permissions per integration" -Ak používatelia nemôžu objaviť povrchy protokolov, kvalita prijatia a podpory klesá. +Different clients should have least-privilege access to tool categories. -**Ako to rieši OmniRoute:** +**How OmniRoute solves it:** -- Konsolidovaná stránka**Koncové body**s kartami pre proxy, MCP, A2A a koncové body API -- Inline prepínače stavu služby (Online/Offline) pre MCP a A2A -- Odkazy z prehľadu na špeciálne karty správy
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling - -🧪 27. „Potrebujem komplexné overenie protokolu so skutočnými klientmi“ + -Falošné testy nestačia na overenie kompatibility protokolu pred vydaním. +
+⚙️ 22. "I need operational controls without redeploying" -**Ako to rieši OmniRoute:** +Teams need quick runtime changes during incidents or cost events. -- E2E balík, ktorý spúšťa aplikáciu a využíva skutočný prenos klienta MCP SDK -- Klient A2A testuje toky zisťovania, odosielania, streamovania, získavania a rušenia -- Krížová kontrola tvrdení proti auditu MCP a API úloh A2A
+**How OmniRoute solves it:** - -📡 28. „Potrebujem jednotnú pozorovateľnosť naprieč všetkými rozhraniami“ +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -Rozdelenie pozorovateľnosti podľa protokolu vytvára slepé miesta a dlhšie MTTR. + -**Ako to rieši OmniRoute:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Zjednotené informačné panely / protokoly / analýzy v jednom produkte -- Zdravie + audit + telemetria požiadaviek cez vrstvy OpenAI, MCP a A2A -- Operačné API pre stav a automatizáciu
+Without lifecycle visibility, task incidents become hard to triage. - -💼 29. "Potrebujem jeden runtime pre proxy + nástroje + orchestráciu agenta" +**How OmniRoute solves it:** -Prevádzka mnohých samostatných služieb zvyšuje prevádzkové náklady a spôsoby zlyhania. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**Ako to rieši OmniRoute:** + -- Proxy, server MCP a server A2A kompatibilný s OpenAI v jednom zásobníku -- Zdieľaná autentifikácia, odolnosť, ukladanie údajov a pozorovateľnosť -- Konzistentný model politiky na všetkých interakčných plochách +
+🌊 24. "I need active stream metrics for A2A load" - -🚀 30. „Potrebujem doručiť agentské pracovné postupy bez rozrastania sa kódu lepidla“ +Streaming workflows require operational insight into concurrency and live connections. -Tímy strácajú rýchlosť pri spájaní viacerých ad-hoc služieb a skriptov. +**How OmniRoute solves it:** -**Ako to rieši OmniRoute:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Jednotná stratégia koncových bodov pre klientov a agentov -- Vstavané používateľské rozhrania na správu protokolov a cesty overovania dymu -- Základy pripravené na výrobu (zabezpečenie, protokolovanie, odolnosť, zálohovanie)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Príručka A: Maximalizujte platené predplatné + lacné zálohovanie**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -610,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Príručka B: Balík kódovania s nulovými nákladmi**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Príručka C: 24/7 vždy zapnutý záložný reťazec**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -633,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Príručka D: Operačný program agenta s MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Nastavte kódovanie AI za pár minút za**$0/mesiac**. Pripojte tieto bezplatné účty a použite vstavanú kombináciu**Free Stack**. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Krok | Akcia | Poskytovatelia odblokovaní | -| ---- | --------------------------------------------------- | ------------------------------------------------------------------- | -| 1 | Pripojiť**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**neobmedzene**| -| 2 | Pripojte**Qoder**(Google OAuth) | kimi-k2-myslenie, qwen3-coder-plus, deepseek-r1... —**neobmedzené**| -| 3 | Pripojte**Qwen**(kód zariadenia) | qwen3-coder-plus, qwen3-coder-flash... —**neobmedzene**| -| 4 | Pripojte**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180 000/mesiac zadarmo**| -| 5 | `/dashboard/combos` → Šablóna**Free Stack (0 $)**| Round-robin všetkých bezplatných poskytovateľov automaticky | +| Step | Action | Providers Unlocked | +| ---- | -------------------------------------------------- | ------------------------------------------------------------------ | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Nasmerujte ľubovoľné IDE/CLI na:**`http://localhost:20128/v1` · Kľúč API: `any-string` · Hotovo. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Voliteľné extra pokrytie (tiež zadarmo):**Groq API kľúč (30 RPM zadarmo), NVIDIA NIM (40 RPM zadarmo, 70+ modelov), Cerebras (1 milión tokenov/deň), LongCat API kľúč (50 miliónov tokenov/deň!), Cloudflare Workers AI (10 000 neurónov/deň, 50+ modelov).## Rýchly štart +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Rýchly štart ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **Používatelia pnpm:**Po inštalácii spustite príkaz `pnpm accept-builds -g`, aby ste povolili natívne skripty zostavovania vyžadované `better-sqlite3` a `@swc/core`: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > > ```bash > pnpm install -g omniroute -> pnpm schváliť-builds -g # Vybrať všetky balíky → schváliť +> pnpm approve-builds -g # Select all packages → approve > omniroute > ``` -Dashboard sa otvorí na `http://localhost:20128` a základná adresa URL API je `http://localhost:20128/v1`. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Príkaz | Popis | -| ----------------------- | ----------------------------------------------------------------- | -| "omniroute" | Spustite server (`PORT=20128`, API a dashboard na rovnakom porte) | -| `omniroute --port 3000` | Nastavte kanonický/API port na 3000 | -| `omniroute --mcp` | Spustiť MCP server (stdio transport) | -| `omniroute --no-open` | Neotvárať automaticky prehliadač | -| `omniroute --help` | Zobraziť pomoc | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Voliteľný režim s rozdeleným portom:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -Pre väčšinu nasadení potrebujete iba: +For most deployments, you only need: -| Premenná | Predvolené | Účel | -| ------------------------- | ------------------------------ | ------------------------------------------------------------- -------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | "600 000" | Zdieľaná základná línia pre upstream načítanie, skryté časové limity Undici, požiadavky na odtlačky prstov TLS a časové limity žiadostí o premostenie API/proxy | -| `STREAM_IDLE_TIMEOUT_MS` | zdedí `REQUEST_TIMEOUT_MS` | Maximálna medzera medzi streamovanými časťami predtým, ako OmniRoute preruší stream SSE | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -Spätná kompatibilita je zachovaná: existujúce premenné „FETCH_TIMEOUT_MS“, „API_BRIDGE_PROXY_TIMEOUT_MS“ a ďalšie premenné časového limitu pre jednotlivé vrstvy stále fungujú a prepisujú zdieľanú základnú líniu. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Ak potrebujete jemnejšie ovládanie, sú k dispozícii pokročilé prepísania:| Premenná | Predvolené | Účel | -| ----------------------------------------- | ------------------------------------------- | --------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | zdedí `REQUEST_TIMEOUT_MS` | Celkový časový limit upstream požiadavky použitý hlavným signálom prerušenia vyzdvihnutia | -| `FETCH_HEADERS_TIMEOUT_MS` | zdedí `FETCH_TIMEOUT_MS` | Undici časový limit na prijímanie hlavičiek odozvy smerom hore | -| `FETCH_BODY_TIMEOUT_MS` | zdedí `FETCH_TIMEOUT_MS` | Undici časový limit medzi blokmi tela upstream (`0` ho zakáže) | -| `FETCH_CONNECT_TIMEOUT_MS` | "30 000" | Časový limit pripojenia Undici TCP | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | "4000" | Undici nečinný keep-alive socket timeout | -| `TLS_CLIENT_TIMEOUT_MS` | zdedí `FETCH_TIMEOUT_MS` | Časový limit pre žiadosti o odtlačky prstov TLS uskutočnené prostredníctvom `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | zdedí `REQUEST_TIMEOUT_MS` alebo `30000` | Časový limit pre presmerovanie proxy servera `/v1` z portu API na port dashboardu | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300 000)` | Časový limit prichádzajúcej požiadavky na serveri API mosta | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | "60 000" | Časový limit prichádzajúcej hlavičky na serveri mosta API | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | 5000 | Časový limit udržiavania na serveri mosta API | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | "0" | Časový limit nečinnosti soketu na serveri API mosta (`0` ho zakáže) | +Advanced overrides are available if you need finer control: -Ak spúšťate OmniRoute za Nginx, Caddy, Cloudflare alebo iným reverzným proxy serverom, uistite sa, že -časové limity sú tiež vyššie ako časové limity vášho streamu/načítania OmniRoute.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Otvorte Dashboard → `Providers` a pripojte aspoň jedného poskytovateľa (OAuth alebo API kľúč). -2. Otvorte Dashboard → `Koncové body` a vytvorte kľúč API. -3. (Voliteľné) Otvorte Dashboard → `Combos` a nastavte svoj záložný reťazec.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Funguje s Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode a SDK kompatibilné s OpenAI.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (pre operácie poháňané nástrojmi):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -Potom pripojte svojho klienta MCP cez `stdio` a otestujte nástroje, ako sú: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` -- `kombiny_zoznamu_všetkých_cestov` +- `omniroute_list_combos` -**A2A (pre pracovné postupy agent-agent):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -762,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Tento balík overuje skutočné toky klientov MCP a A2A oproti spustenej aplikácii.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -770,13 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` - -Void Linux (šablóna `xbps-src`) +
+Void Linux (`xbps-src` template) -Pre používateľov Void Linuxu môžete vytvoriť natívny balík pomocou `xbps-src`. Uložte tento blok ako `srcpkgs/omniroute/template`:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -788,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -796,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -872,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -883,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute je k dispozícii ako verejný obrázok Docker na [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Rýchly beh:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -893,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**So súborom prostredia:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Pomocou Docker Compose:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Podpora panelov pre nasadenia Docker teraz zahŕňa**Cloudflare Quick Tunnel**na jedno kliknutie na `Dashboard → Endpoints`. Prvý povolí sťahovanie `cloudflared` iba v prípade potreby, spustí dočasný tunel k vášmu aktuálnemu koncovému bodu `/v1` a zobrazí vygenerovanú URL `https://*.trycloudflare.com/v1` priamo pod vašou normálnou verejnou adresou URL. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Poznámky: +Notes: -- Adresy URL rýchleho tunela sú dočasné a menia sa po každom reštarte. -- Rýchle tunely sa po reštarte OmniRoute alebo kontajnera automaticky neobnovia. V prípade potreby ich znova povoľte z palubnej dosky. -- Spravovaná inštalácia momentálne podporuje Linux, macOS a Windows na `x64` / `arm64`. -- Spravované rýchle tunely predvolene používajú prenos HTTP/2, aby sa predišlo hlučným upozorneniam vyrovnávacej pamäte QUIC UDP v prostredí s obmedzenými kontajnermi. Ak chcete iný prenos, nastavte `CLOUDFLARED_PROTOCOL=quic` alebo `auto`. -- Obrázky Docker spájajú korene systémovej CA a odovzdávajú ich spravovanému „cloudflared“, čo zabraňuje zlyhaniam dôvery TLS pri zavádzaní tunela vo vnútri kontajnera. -- SQLite beží v režime WAL. `docker stop` by sa malo nechať dokončiť, aby OmniRoute mohol skontrolovať najnovšie zmeny späť do `storage.sqlite`. -- V pribalených súboroch Compose je už nastavená doba odkladu 40 sekúnd. Ak spustíte obraz priamo, ponechajte `--stop-timeout 40` (alebo podobne), aby manuálne zastavenia neprerušili čistenie pri vypnutí. -- Nastavte `CLOUDFLARED_BIN=/absolútna/cesta/k/cloudflared`, ak chcete, aby OmniRoute namiesto sťahovania používal existujúci binárny súbor. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Používanie Docker Compose s Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute možno bezpečne odkryť pomocou automatického poskytovania SSL Caddy. Uistite sa, že záznam DNS A vašej domény ukazuje na IP adresu vášho servera.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` +| Image | Tag | Size | Description | +| ------------------------ | -------- | ------ | --------------------- | +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | -| Obrázok | Tag | Veľkosť | Popis | -| ------------------------- | -------- | ------ | ---------------------- | -| `diegosouzapw/omniroute` | "najnovšie" | ~250 MB | Najnovšie stabilné vydanie | -| `diegosouzapw/omniroute` | "1.0.3" | ~250 MB | Aktuálna verzia |--- +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**NOVINKA!**OmniRoute je teraz k dispozícii ako**natívna desktopová aplikácia**pre Windows, macOS a Linux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Spustite OmniRoute ako samostatnú počítačovú aplikáciu – pre miestne modely nie je potrebný žiadny terminál, žiadny prehliadač ani internet. Aplikácia založená na elektróne obsahuje: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Natívne okno**— Vyhradené okno aplikácie s integráciou do systémovej lišty -- 🔄**Auto-Start**– Spustite OmniRoute pri prihlásení do systému -- 🔔**Natívne upozornenia**– Získajte upozornenia na vyčerpanie kvóty alebo problémy s poskytovateľom -- ⚡**Inštalácia jedným kliknutím**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Režim offline**– Funguje úplne offline s pribaleným serverom### Rýchly štart +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Rýchly štart ```bash # Development mode @@ -982,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -Keď je minimalizovaný, OmniRoute žije vo vašej systémovej lište s rýchlymi akciami: +When minimized, OmniRoute lives in your system tray with quick actions: -- Otvorte palubnú dosku -- Zmeňte port servera -- Ukončite aplikáciu +- Open dashboard +- Change server port +- Quit application -📖 Úplná dokumentácia: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Úroveň | Poskytovateľ | Náklady | Obnovenie kvóty | Najlepšie pre | -| ----------------- | --------------------------- | ----------------------------------- | ---------------------------- | -------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 PREDPLATNÉ** | Claude Code (Pro) | 20 USD/mesiac | 5h + týždenne | Už prihlásené | -| | Codex (Plus/Pro) | 20 – 200 USD/mesiac | 5h + týždenne | Používatelia OpenAI | -| | Gemini CLI | **ZADARMO** | 180 tis./mesiac + 1 tis./deň | Všetci! | -| | GitHub Copilot | 10 – 19 USD/mes. | Mesačne | Používatelia GitHubu | -| **🔑 API KEY** | NVIDIA NIM | **ZADARMO**(dev forever) | ~40 RPM | 70+ otvorených modelov | -| | Cerebras | **ZADARMO**(1 milión toku/deň) | 60 000 TPM / 30 RPM | Najrýchlejší na svete | -| | Groq | **ZADARMO**(30 RPM) | 14,4K RPD | Ultra rýchla lama/gemma | -| | DeepSeek V3.2 | 0,27 USD/1,10 USD za 1 milión | Žiadne | Najlepšie zdôvodnenie cena/kvalita | -| | xAI Grok-4 Fast | **0,20 USD/0,50 USD za 1M**🆕 | Žiadne | Najrýchlejšie + volanie nástroja, ultranízke | -| | xAI Grok-4 (štandard) | 0,20 USD/1,50 USD za 1 milión 🆕 | Žiadne | Rozumná vlajková loď od xAI | -| | Mistral | Bezplatná skúšobná verzia + platené | Obmedzená sadzba | Európska AI | -| | OpenRouter | Platba za použitie | Žiadne | 100+ modelov agr. | -| **💰 LACNO** | GLM-5 (cez Z.AI) 🆕 | 0,5 USD/1 milión | Denne 10:00 | 128K výstup, najnovšia vlajková loď | -| | GLM-4,7 | 0,6 USD/1 milión | Denne 10:00 | Záloha rozpočtu | -| | MiniMax M2,5 🆕 | Vstup 0,3 $/1 milión | 5-hodinové valcovanie | Úvahy + agentské úlohy | -| | MiniMax M2.1 | 0,2 USD/1 milión | 5-hodinové valcovanie | Najlacnejšia možnosť | -| | Kimi K2.5 (Moonshot API) 🆕 | Platba za použitie | Žiadne | Priamy prístup Moonshot API | -| | Kimi K2 | 9 USD/mesiac byt | 10 miliónov tokenov/mesiac | Predvídateľné náklady | -| **🆓 ZDARMA** | Qoder | **$0** | Neobmedzené | 5 modelov neobmedzene | -| | Qwen | **$0** | Neobmedzené | 4 modely neobmedzene | -| | Kiro | **$0** | Neobmedzené | Claude Sonnet/Haiku (staviteľ AWS) | -| | LongCat Flash-Lite 🆕 | **$0**(50 miliónov tok/deň 🔥) | 1 RPS | Najväčšia bezplatná kvóta na Zemi | -| | Opeľovanie AI 🆕 | **$0**(nie je potrebný žiadny kľúč) | 1 požiadavka/15s | GPT-5, Claude, DeepSeek, Llama 4 | -| | Cloudflare Workers AI 🆕 | **$0**(10 000 neurónov/deň) | ~150 resp./deň | 50+ modelov, globálny náskok | -| | Scaleway AI 🆕 | **$0**(celkom 1 milión tokenov) | Obmedzená sadzba | EU/GDPR, Qwen3 235B, Lama 70B | > 🆕**Pridané nové modely (marec 2026):**Rodina Grok-4 Fast za 0,20 $/0,50 $/M (porovnávacia rýchlosť 1143 ms – o 30 % rýchlejšia ako Gemini 2.5 Flash), GLM-5 cez Z.AI s výstupom 128 kB, aktualizovaná cena MiniMax M2.5 V5, Kimi K2.2 Direct API. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 Combo Stack 0 $ — Kompletné bezplatné nastavenie:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Nulové náklady. Kódovanie sa nikdy nezastaví.**Nakonfigurujte si to ako jednu kombináciu OmniRoute a všetky núdzové situácie sa stanú automaticky – žiadne manuálne prepínanie.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Všetky modely uvedené nižšie sú**100 % zadarmo, bez potreby kreditnej karty**. OmniRoute medzi nimi automaticky prechádza, keď sa minie jedna kvóta – skombinujte ich všetky a získate nerozbitnú kombináciu 0 $.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Model | Predpona | Limit | Limit sadzby | -| -------------------- | ------ | -------------- | ---------------------- | -| `claude-sonnet-4.5` | `kr/` |**Neobmedzené**| Žiadny hlásený denný limit | -| `claude-haiku-4,5` | `kr/` |**Neobmedzené**| Žiadny hlásený denný limit | -| `claude-opus-4.6` | `kr/` |**Neobmedzené**| Najnovší Opus cez Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) -| Model | Predpona | Limit | Limit sadzby | -| ------------------- | ------ | -------------- | ---------------- | -| "kimi-k2-myslenie" | „ak/“ |**Neobmedzené**| Žiadny nahlásený strop | -| "qwen3-coder-plus" | „ak/“ |**Neobmedzené**| Žiadny nahlásený strop | -| `deepseek-r1` | „ak/“ |**Neobmedzené**| Žiadny nahlásený strop | -| "minimax-m2,1" | „ak/“ |**Neobmedzené**| Žiadny nahlásený strop | -| "kimi-k2" | „ak/“ |**Neobmedzené**| Žiadny nahlásený strop | +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | --------------------- | +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -> Odporúčaný spôsob pripojenia:**Token osobného prístupu + `qodercli`**. OAuth prehliadača je -> experimentálne a predvolene vypnuté, pokiaľ nie sú nakonfigurované premenné prostredia `QODER_OAUTH_*`.### 🟡 QWEN MODELS (Device Code Auth) +### 🟢 QODER MODELS (Free PAT via qodercli) -| Model | Predpona | Limit | Limit sadzby | -| -------------------- | ------ | -------------- | -------------------- | -| "qwen3-coder-plus" | `qw/` |**Neobmedzené**| Žiadny nahlásený strop | -| "qwen3-coder-flash" | `qw/` |**Neobmedzené**| Žiadny nahlásený strop | -| `qwen3-coder-next` | `qw/` |**Neobmedzené**| Žiadny nahlásený strop | -| "model vízie" | `qw/` |**Neobmedzené**| Multimodálne (obrázky) |### 🟣 GEMINI CLI (Google OAuth) +| Model | Prefix | Limit | Rate Limit | +| ------------------ | ------ | ------------- | --------------- | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -| Model | Predpona | Limit | Limit sadzby | -| ------------------------- | ------ | ---------------------------- | -------------- | -| `gemini-3-flash-preview` | `gc/` |**180 tis. tok/mesiac**+ 1 tis./deň | Mesačný reset | -| "gemini-2,5-pro" | `gc/` | 180 tis./mesiac (zdieľaný bazén) | Vysoká kvalita |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| Úroveň | Denný limit | Limit sadzby | Poznámky | +### 🟡 QWEN MODELS (Device Code Auth) + +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | ------------------- | +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | + +### 🟣 GEMINI CLI (Google OAuth) + +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | + +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---------- | ------------ | ----------- | ------------------------------------------------------ | -| Zadarmo (Dev) | Žiadny token cap |**~40 RPM**| 70+ modelov; prechod na limity čistej sadzby v polovici roku 2025 | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -Populárne bezplatné modely: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-epser `destructekse`,destructor `destructekse`1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -| Úroveň | Denný limit | Limit sadzby | Poznámky | -| ---- | ------------------ | ----------------- | -------------------------------------------- | -| Zadarmo |**1 milión tokenov/deň**| 60 000 TPM / 30 RPM | Svetovo najrýchlejšie odvodenie LLM; resetuje sa denne | +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -Dostupné zadarmo: `lama-3.3-70b`, `lama-3.1-8b`, `deepseek-r1-distill-lama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -| Úroveň | Denný limit | Limit sadzby | Poznámky | -| ---- | -------------- | ----------------- | ------------------------------------------ | -| Zadarmo |**14,4K RPD**| 30 otáčok za minútu na model | Žiadna kreditná karta; 429 na limit, neúčtuje sa | +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -Dostupné zadarmo: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +### 🔴 GROQ (Free API Key — console.groq.com) -| Model | Predpona | Denná bezplatná kvóta | Poznámky | -| ------------------------------ | ------ | ------------------ | ------------------------ | -| "LongCat-Flash-Lite" | `lc/` |**50 miliónov žetónov**💥 | Najväčšia bezplatná kvóta vôbec | -| "LongCat-Flash-Chat" | `lc/` | 500 000 žetónov | Viacotáčkový chat | -| "LongCat-Flash-Thinking" | `lc/` | 500 000 žetónov | Zdôvodnenie / CoT | -| "LongCat-Flash-Thinking-2601" | `lc/` | 500 000 žetónov | Verzia z januára 2026 | -| "LongCat-Flash-Omni-2603" | `lc/` | 500 000 žetónov | Multimodálne | +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ------------- | ---------------- | ----------------------------------------- | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -> 100 % zadarmo vo verejnej beta verzii. Zaregistrujte sa na [longcat.chat](https://longcat.chat) pomocou e-mailu alebo telefónu. Resetuje sa denne o 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -| Model | Predpona | Limit sadzby | Poskytovateľ za | -| ---------- | ------ | ---------- | ------------------- | -| "openai" | `pol/` | 1 požiadavka/15s | GPT-5 | -| "claude" | `pol/` | 1 požiadavka/15s | Antropický Claude | -| "blíženec" | `pol/` | 1 požiadavka/15s | Google Gemini | -| "hlboké vyhľadávanie" | `pol/` | 1 požiadavka/15s | DeepSeek V3 | -| "lama" | `pol/` | 1 požiadavka/15s | Meta Llama 4 Scout | -| "mistral" | `pol/` | 1 požiadavka/15s | Mistral AI | +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 -> ✨**Nulové trenie:**Bez registrácie, bez kľúča API. Pridajte poskytovateľa Pollinations s prázdnym kľúčovým poľom a funguje to okamžite.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | -| Úroveň | Denné neuróny | Ekvivalentné použitie | Poznámky | -| ---- | -------------- | ---------------------------------------- | ------------------------ | -| Zadarmo |**10 000**| ~150 LLM resp / 500 s audio / 15 000 vložení | Global edge, 50+ modelov | +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. -Populárne bezplatné modely: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (bezplatný zvuk!), `@cf/qwen/qwen2.5-coder-`1 +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 -> Vyžaduje token API + ID účtu z [dash.cloudflare.com](https://dash.cloudflare.com). Uložte ID účtu v nastaveniach poskytovateľa.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +| Model | Prefix | Rate Limit | Provider Behind | +| ---------- | ------ | ---------- | ------------------ | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -| Úroveň | Voľná ​​kvóta | Miesto | Poznámky | -| ---- | -------------- | ------------ | ------------------------------------ | -| Zadarmo |**1 milión tokenov**| 🇫🇷 Paríž, EÚ | V rámci limitov nie je potrebná žiadna kreditná karta | +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -Dostupné zadarmo: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 -> V súlade s EÚ/GDPR. Získajte kľúč API na [console.scaleway.com](https://console.scaleway.com). +| Tier | Daily Neurons | Equivalent Usage | Notes | +| ---- | ------------- | --------------------------------------- | ----------------------- | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | ->**💡 Konečný bezplatný balík (11 poskytovateľov, 0 $ navždy):** +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` + +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. + +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | +| ---- | ------------- | ------------ | ----------------------------------- | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | + +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` + +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). + +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Sonnet/Haiku NEOBMEDZENÉ -> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -> LongCat Lite (lc/) → LongCat-Flash-Lite – 50 miliónov tokenov/deň 🔥 -> Opeľovanie (pol/) → GPT-5, Claude, DeepSeek, Llama 4 – nie je potrebný žiadny kľúč -> Qwen (qw/) → qwen3-kódovacie modely NEOBMEDZENÉ -> Gemini (gemini/) → Gemini 2.5 Flash – 1 500 req/deň zadarmo -> Cloudflare AI (cf/) → 50+ modelov – 10 000 neurónov/deň -> Scaleway (scw/) → Qwen3 235B, Llama 70B – 1 milión bezplatných tokenov (EÚ) -> Groq (groq/) → Llama/Gemma – ultrarýchle 14,4 000 req/deň -> NVIDIA NIM (nvidia/) → 70+ otvorených modelov — 40 RPM navždy -> Cerebras (cerebras/) → Najrýchlejšie lamy/Qwen na svete – 1 milión tok/deň -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Prepíšte akýkoľvek zvuk/video za**$0**– Deepgram vedie s 200 $ zadarmo, AssemblyAI 50 $ záložný, Groq Whisper ako neobmedzená núdzová záloha. +## 🎙️ Free Transcription Combo -| Poskytovateľ | Bezplatné kredity | Najlepší model | Limit sadzby | -| ------------------ | ----------------------- | --------------------------------------------- | ----------------------------- | -| 🢢**Deepgram**|**200 $ zadarmo**(registrácia) | `nova-3` — najlepšia presnosť, viac ako 30 jazykov | Žiadny limit RPM na bezplatné kredity | -| 🔵**MontážAI**|**50 $ zadarmo**(registrácia) | "universal-3-pro" – kapitoly, sentiment, PII | Žiadny limit RPM na bezplatné kredity | -| 🔴**Groq**|**Navždy zadarmo**| `whisper-large-v3` — OpenAI Whisper | 30 RPM (obmedzená rýchlosť) | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. -**Odporúčaná kombinácia v `/dashboard/combos`:**``` +| Provider | Free Credits | Best Model | Rate Limit | +| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | + +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Potom na `/dashboard/media` → karta**Prepis**: nahrajte akýkoľvek zvukový alebo video súbor → vyberte svoj kombinovaný koncový bod → získajte prepis v podporovaných formátoch.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 je postavený ako operačná platforma, nie len ako relé proxy.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Funkcia | Čo to robí | -| ---------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------ | -| ⚡**Grok-4 Fast Family** | Modely xAI za 0,20 USD/0,50 USD/M – porovnávacia rýchlosť 1143 ms (o 30 % rýchlejšie ako Gemini 2.5 Flash) | -| 🧠**GLM-5 cez Z.AI** | 128 000 výstupný kontext, 0,5 $/1 milión – najnovšia vlajková loď z rodiny GLM | -| 🔮**MiniMax M2,5** | Úvahy + agentské úlohy za 0,30 $/1 milión – významný upgrade z M2,1 | -| 🎯**toolCalling Flag na model** | Pre model „toolCalling: true/false“ v registri – AutoCombo preskočí modely bez nástrojov | -| 🌍**Multilingual Intent Detection** | Kľúčové slová PT/ZH/ES/AR v hodnotení AutoCombo – lepší výber modelu pre neanglický obsah | -| 📊**Návraty založené na benchmarkoch** | Skutočná latencia p95 z kombinovaného hodnotenia informačných kanálov živých žiadostí – AutoCombo sa učí zo skutočných údajov | -| 🔁**Požiadať o deduplikáciu** | Okno odstraňovania duplicitných údajov založené na hašovaní obsahu – bezpečné pre viacerých agentov, zabraňuje duplicitným poplatkom | -| 🔌**Stratégia pripojiteľného smerovača** | Rozšíriteľné rozhranie `RouterStrategy` – pridajte vlastnú logiku smerovania ako zásuvné moduly | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Funkcia | Čo to robí | -| --------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Modelové ihrisko** | Stránka hlavného panela na priame otestovanie akéhokoľvek modelu – selektory poskytovateľa/modelu/koncového bodu, editor Monaco, streamovanie, prerušenie, načasovanie | -| 🔏**CLI Fingerprint Matching** | Usporiadanie hlavičky/tela podľa poskytovateľa tak, aby sa zhodovalo s natívnymi podpismi CLI – prepínajte podľa poskytovateľa v Nastavenia > Zabezpečenie.**Vaša proxy IP je zachovaná** | -| 🤝**Podpora ACP (Protokol agenta klienta)** | Objavovanie agentov CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 ďalších), spúšťač procesov, koncový bod `/api/acp/agents` | -| 🤖**Hlavný panel agentov AKT** | Debug › Stránka Agenti — mriežka 14 agentov so stavom inštalácie, verziou, vlastným formulárom agenta pre ľubovoľný nástroj CLI. Používatelia**OpenCode**dostanú tlačidlo „Stiahnuť opencode.json“, ktoré automaticky vygeneruje konfiguráciu pripravenú na použitie so všetkými dostupnými modelmi. | -| 🔧**Smerovanie vlastného modelu `apiFormat`** | Vlastné modely s `apiFormat: "responses"` teraz správne smerujú do prekladača Responses API | -| 🏢**Izolácia pracovného priestoru Codex** | Viaceré pracovné priestory Codex na e-mail – OAuth správne oddeľuje pripojenia podľa ID pracovného priestoru | -| 🔄**Elektrónová automatická aktualizácia** | Desktopová aplikácia kontroluje aktualizácie + automatická inštalácia pri reštarte | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Funkcia | Čo to robí | -| --------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**MCP Server (25 nástrojov)** | Nástroje IDE/agenta cez 3 prenosy: stdio, SSE (`/api/mcp/sse`), Streamovateľný HTTP (`/api/mcp/stream`). 18 jadier + 3 pamäte + 4 nástroje zručností | -| 🤝**Server A2A (JSON-RPC + SSE)** | Vykonávanie úloh agent-agent so synchronizáciou a streamovaním | -| 🧭**Stránka konsolidovaných koncových bodov** | Stránka správy s kartami s kartami Endpoint Proxy, MCP, A2A a API Endpoints | -| 🎚️**Prepínače aktivácie/deaktivácie služby** | Vypínače ON/OFF pre MCP a A2A s trvalým nastavením (predvolené: OFF) | -| 🛰️**MCP Runtime Heartbeat** | Skutočný stav procesu (pid, uptime, srdcový tep, transport, režim rozsahu) | -| 📋**MCP Audit Trail** | Filtrovateľné protokoly auditu s úspechom/neúspechom a kľúčovým priradením | -| 🔐**Presadzovanie rozsahu MCP** | 10 podrobných povolení rozsahu pre riadený prístup k nástroju | -| 📡**Správa životného cyklu úloh A2A** | Zoznam/filtrovanie úloh, kontrola udalostí/artefaktov, zrušenie spustených úloh | -| 📋**Zistenie karty agenta** | `/.well-known/agent.json` pre automatické zisťovanie klienta | -| 🧪**Protokol E2E Test Harness** | Skutočný klient MCP SDK + A2A toky v `test:protocols:e2e` | -| ⚙️**Ovládacie prvky** | Kombinované prepínanie, aplikovanie profilov odolnosti, resetovanie ističov z jedného ovládacieho povrchu | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Funkcia | Čo to robí | -| -------------------------------------------- | ---------------------------------------------------------------------------------- | ----------------------- | -| 🎯**Inteligentná 4-úrovňová núdzová záloha** | Automatická trasa: Predplatné → Kľúč API → Lacné → Zadarmo | -| 📊**Sledovanie kvóty v reálnom čase** | Živý počet tokenov + reset odpočítavania na poskytovateľa | -| 🔄**Formátový preklad** | OpenAI ↔ Claude ↔ Gemini ↔ Odpovede s konverziami bezpečnými pre schému | -| 👥**Podpora viacerých účtov** | Viac účtov na poskytovateľa s inteligentným výberom | -| 🔄**Automatická obnova tokenov** | Tokeny OAuth sa automaticky obnovia s opätovným pokusom | -| 🎨**Vlastné kombá** | 9 vyvažovacích stratégií + núdzové riadenie reťazca | -| 🌐**Wildcard Router** | `poskytovateľ/*` dynamické smerovanie | -| 🧠**Premýšľajte o kontrolách rozpočtu** | Limity priechodného, ​​automatického, vlastného a adaptívneho uvažovania | -| 🔀**Aliasy modelov** | Vstavaný + vlastný model aliasing a bezpečnosť migrácie | -| ⚡**Degradácia pozadia** | Smerujte úlohy s nízkou prioritou na pozadí na lacnejšie modely | -| 🧪**Inteligentné smerovanie podľa úloh** | Automatický výber modelu podľa typu obsahu (kódovanie/vízia/analýza/sumarizácia) | -| 🔄**Pracovné postupy agentov A2A** | Deterministický FSM orchestrátor pre stavové viackrokové spúšťanie agentov | -| 🔀**Adaptívne smerovanie** | Dynamické prepisovanie stratégie založené na objeme tokenov a zložitosti okamžitej | -| 🎲**Rozmanitosť poskytovateľov** | Shannonovo skóre entropie vyrovnávanie auto-kombo rozloženia dopravy | -| 💬**Promptné vstrekovanie systému** | Dôsledne uplatňované globálne kontroly správania | -| 📄**Kompatibilita rozhrania Responses API** | Plná podpora `/v1/responses` pre kódex a pokročilé pracovné postupy agentov | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Funkcia | Čo to robí | -| ---------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ---------------------------------------- | -| 🖼️**Generovanie obrázkov** | `/v1/images/generations` s cloudom a lokálnymi backendmi | -| 📐**Vloženie** | `/v1/embeddings` pre vyhľadávanie a RAG potrubia | -| 🎤**Prepis zvuku** | `/v1/audio/transscriptions` — 7 poskytovateľov (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), automatická detekcia jazyka, podpora MP4/MP3/WAV | -| 🔊**Prevod textu na reč** | `/v1/audio/speech` — 10 poskytovateľov (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) so správnymi chybovými hláseniami | -| 🎬**Generácia videa** | `/v1/videos/generations` (pracovné postupy ComfyUI + SD WebUI) | -| 🎵**Hudobná generácia** | `/v1/music/generations` (pracovné postupy ComfyUI) | -| 🛡️**Moderovania** | `/v1/moderations` bezpečnostné kontroly | -| 🔀**Reranking** | `/v1/rerank` pre hodnotenie relevantnosti | -| 🔍**Vyhľadávanie na webe**🆕 | `/v1/search` — 5 poskytovateľov (Serper, Brave, Perplexity, Exa, Tavily), 6 500+ zadarmo/mesiac, automatické zlyhanie, vyrovnávacia pamäť | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Funkcia | Čo to robí | -| ----------------------------------------------- | ------------------------------------------------------------------------------------------------------ | -------------------------------- | -| 🔌**Ističe** | Vypnutie/obnovenie pre každý model s prahovými ovládačmi | -| 🎯**Modely s vedomím koncového bodu** | Vlastné modely deklarujú podporované koncové body + formát API | -| 🛡️**Anti-hromové stádo** | Mutex + semaforové ochrany pri opakovaných udalostiach/postupoch | -| 🧠**Sémantická + podpisová vyrovnávacia pamäť** | Zníženie nákladov/latencie pomocou dvoch vrstiev vyrovnávacej pamäte | -| ⚡**Požiadajte o idempotenciu** | Duplicitné ochranné okno | -| 🔒**Spoofing odtlačkov prstov TLS** | Odtlačok TLS podobný prehliadaču –**obmedzuje detekciu robotov a označovanie účtu** | -| 🔏**CLI Fingerprint Matching** | Zodpovedá natívnym podpisom požiadaviek CLI –**znižuje riziko zákazu pri zachovaní proxy IP** | -| 🌐**Filtrovanie IP** | Kontrola zoznamu povolených/blokovaných nasadení pre vystavené nasadenia | -| 📊**Upraviteľné limity sadzieb** | Konfigurovateľné globálne limity/limity na úrovni poskytovateľa s perzistenciou | -| 📉**Pôvabná degradácia** | Viacvrstvové záložné funkcie chrániace operácie základnej brány | -| 📜**Config Audit Trail** | Sledovanie zmeny založené na rozdieloch, ktoré bráni prevádzkovému posunu jednoduchým vrátením späť | -| ⏳**Provider Health Sync** | Proaktívne monitorovanie uplynutia platnosti tokenu, ktoré spúšťa výstrahy pred zlyhaniami autorizácie | -| 🚪**Automaticky zakázať zakázané účty** | Prevádzkový istič automaticky zaplombuje trvalo zablokované tokenové účty | -| 🔑**Správa kľúčov API + rozsah** | Bezpečné vydávanie/otočenie kľúčov a ovládanie modelu/poskytovateľa | -| 👁️**Scoped API Key Reveal**🆕 | Obnovenie kľúčov API pomocou možnosti `ALLOW_API_KEY_REVEAL` | -| 🛡️**Chránené `/modely`** | Voliteľné overenie a skrytie poskytovateľa pre katalóg modelov | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Funkcia | Čo to robí | -| ------------------------------------ | ---------------------------------------------------------------------------------------- | ---------------------------- | -| 📝**Žiadosť + protokolovanie proxy** | Úplná žiadosť/odpoveď a protokolovanie proxy | -| 📉**Streamované podrobné záznamy**🆕 | Čisto rekonštruuje toky užitočného zaťaženia SSE do používateľského rozhrania | -| 📋**Unified Logs Dashboard** | Požiadavka, proxy, audit a zobrazenie konzoly na jednej stránke | -| 🔍**Požiadať o telemetriu** | p50/p95/p99 latencia a sledovanie požiadaviek | -| 🏥**Panel zdravia** | Doba prevádzkyschopnosti, stavy prerušovačov, uzamknutia, štatistiky vyrovnávacej pamäte | -| 💰**Sledovanie nákladov** | Kontroly rozpočtu a viditeľnosť cien podľa modelu | -| 📈**Analytické vizualizácie** | Štatistiky používania modelu/poskytovateľa a zobrazenia trendov | -| 🧪**Rámec hodnotenia** | Testovanie zlatej sady s konfigurovateľnými stratégiami zhody | -| 📡**Diagnostika naživo**🆕 | Sémantické vynechanie vyrovnávacej pamäte pre presné kombinované živé testovanie | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Funkcia | Čo to robí | -| ---------------------------------------- | ------------------------------------------------------------------------------------------- | --------------------- | -| 🌐**Nasadenie kdekoľvek** | Localhost, VPS, Docker, Cloud prostredia | -| 🚇**Cloudflare Tunel**🆕 | Integrácia rýchleho tunela jedným kliknutím z ovládacieho panela | -| 🔑**Filtrovanie modelu kľúča API** | Natívna odpoveď /v1/modelov filtrovaná prostredníctvom priradených kontextových rolí nosiča | -| ⚡**Smart Cache Bypass** | Konfigurovateľná heuristika TTL a ovládacie prvky núteného opätovného načítania | -| 🔄**Zálohovanie/Obnova** | Export/import a toky obnovy po havárii | -| 🧙**Sprievodca onboardingom** | Sprievodca nastavením pri prvom spustení | -| 🔧**CLI Tools Dashboard** | Nastavenie jedným kliknutím pre obľúbené nástroje kódovania | -| 🎮**Modelové ihrisko** | Otestujte akéhokoľvek poskytovateľa/model/koncový bod z ovládacieho panela | -| 🔏**CLI Fingerprint Toggle** | Zhoda odtlačkov prstov jednotlivých poskytovateľov v časti Nastavenia > Zabezpečenie | -| 🌐**i18n (30 jazykov)** | Úplná podpora dashboardu + dokumentov s podporou RTL | -| 🧹**Vymazať všetky modely** | Vymazanie zoznamu modelov jedným kliknutím v detailoch poskytovateľa | -| 👁️**Ovládacie prvky na bočnom paneli**🆕 | Skryť komponenty a integrácie z Nastavenia vzhľadu | -| 📋**Šablóny vydaní** | Štandardizované šablóny GitHub pre chyby a funkcie | -| 📂**Custom Data Directory** | Prepísanie `DATA_DIR` pre umiestnenie úložiska | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1295,103 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -Keď kvóta, sadzba alebo stav zlyhajú, OmniRoute automaticky prejde na ďalšieho kandidáta bez manuálneho prepínania.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A sú viditeľné v používateľskom rozhraní a dokumentoch (nie sú skryté) -- Rozhrania API stavu protokolu odhaľujú aktuálne prevádzkové údaje (`/api/mcp/*`, `/api/a2a/*`) -- Panely obsahujú akcie pre 2. deň (prepínanie komb, resetovanie prerušovača, zrušenie úlohy)#### Translator + validation workflow +#### Protocol management that is visible and operable -Oblasť prekladateľov zahŕňa: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Ihrisko**: vyžiadajte si kontroly transformácie -**Chat Tester**: úplná spiatočná cesta so žiadosťou a odpoveďou -**Testovacia lavica**: viacero prípadov v jednom behu -**Live Monitor**: zobrazenie premávky v reálnom čase +#### Translator + validation workflow -Plus validácia protokolu so skutočnými klientmi cez `npm run test:protocols:e2e`. +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Odkaz na nástroj, konfigurácie IDE a príklady klientov +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A Server README](src/lib/a2a/README.md)**— Zručnosti, metódy JSON-RPC, streamovanie a životný cyklus úloh## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute obsahuje vstavaný hodnotiaci rámec na testovanie kvality odozvy LLM oproti zlatému súboru. Prístup k nej získate cez**Analytics → Evals**na hlavnom paneli.### Built-in Golden Set +## 🧪 Evaluations (Evals) -Predinštalovaná sada „OmniRoute Golden Set“ obsahuje testovacie prípady pre: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Pozdravy, matematika, geografia, generovanie kódu -- Súlad s formátom JSON, preklad, generovanie markdownov -- Bezpečnostné odmietnutie (škodlivý obsah), počítanie, booleovská logika### Evaluation Strategies +### Built-in Golden Set -| Stratégia | Popis | Príklad | -| ----------------- | ---------------------------------------------------------------------- | ------------------------------- | --- | -| "presne" | Výstup sa musí presne zhodovať | "4"" | -| "obsahuje" | Výstup musí obsahovať podreťazec (nerozlišujú sa malé a veľké písmená) | "Paríž" | -| "regulárny výraz" | Výstup musí zodpovedať vzoru regulárneho výrazu | `"1.*2.*3"` | -| "vlastné" | Vlastná funkcia JS vracia true/false | `(výstup) => výstup.dĺžka > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) - -🧩 Nastavenie MCP (Model Context Protocol) +
+🧩 MCP Setup (Model Context Protocol) -Spustite prenos MCP v režime stdio:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Odporúčaný postup overenia: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Pripojte svojho klienta MCP cez stdio. -2. Spustite `omniroute_get_health`. -3. Spustite `omniroute_list_combos`. -4. Otvorte `/dashboard/mcp` a potvrďte tep, aktivitu a audit. +Useful APIs for automation: -Užitočné API pre automatizáciu: - -- "GET /api/mcp/status". +- `GET /api/mcp/status` - `GET /api/mcp/tools` - `GET /api/mcp/audit` -- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats` - -🤝 Nastavenie A2A (Agent2Agent) + -Objavte agenta:```bash +
+🤝 A2A Setup (Agent2Agent) + +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Pošlite úlohu:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` +Manage lifecycle: -Správa životného cyklu: - -- "GET /api/a2a/status". +- `GET /api/a2a/status` - `GET /api/a2a/tasks` - `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -Operačné používateľské rozhranie: +Operational UI: -- `/dashboard/a2a` pre pozorovateľnosť úlohy/stavu/prúdu a akcie dymu
+- `/dashboard/a2a` for task/state/stream observability and smoke actions - -🧪 End-to-end validácia protokolu + -Overte oba protokoly so skutočnými klientmi:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Týmto sa overí: +This verifies: -- Pripojenie/zoznam/zavolanie klienta MCP SDK -- A2A objav/odoslanie/streamovanie/získanie/zrušenie -- Krížová kontrola údajov v MCP audit a A2A task management API
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs - -💳 Poskytovatelia predplatného### Claude Code (Pro/Max) + + +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1404,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Tip pre profesionálov:**Používajte Opus na zložité úlohy, Sonnet na rýchlosť. OmniRoute sleduje kvótu na model!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1418,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Každý účet Codex má teraz prepínače pravidiel v `Dashboard -> Providers`: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (ZAP./VYP.): vynúti zásadu prahu 5-hodinového okna. -- `Týždenne` (ZAP/VYP): presadzovanie zásady týždenného limitu okna. -- Prahové správanie: keď povolené okno dosiahne využitie >=90 %, daný účet sa preskočí. -- Rotačné správanie: OmniRoute automaticky smeruje k ďalšiemu oprávnenému účtu Codex. -- Resetovať správanie: po uplynutí času `resetAt` poskytovateľa sa účet automaticky znova stane spôsobilým. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Scenáre: +Scenarios: -- `5h ZAP` + `Týždenne ZAP`: účet sa preskočí, keď ktorékoľvek okno dosiahne prah. -- `5h VYP` + `Týždenne ZAP`: účet môže zablokovať iba týždenné používanie. -- `5h ON` + `Týždenne OFF`: iba 5-hodinové používanie môže zablokovať účet. -- `resetAt` prešiel: účet sa automaticky znova zadá do rotácie (bez manuálneho opätovného zapnutia).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1443,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Najlepšia hodnota:**Obrovská bezplatná úroveň! Použite to pred platenými úrovňami.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1458,71 +1662,91 @@ Models:
- -🔑 Kľúčoví poskytovatelia API### NVIDIA NIM (FREE developer access — 70+ models) +
+🔑 API Key Providers -1. Zaregistrujte sa: [build.nvidia.com](https://build.nvidia.com) -2. Získajte bezplatný kľúč API (vrátane 1 000 kreditov na odvodenie) -3. Ovládací panel → Pridať poskytovateľa → NVIDIA NIM: - - Kľúč API: `nvapi-your-key` +### NVIDIA NIM (FREE developer access — 70+ models) -**Modely:**`nvidia/lama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` a 50+ ďalších +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Tip pre profesionálov:**API kompatibilné s OpenAI – bezproblémovo funguje s prekladom formátu OmniRoute!### DeepSeek +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -1. Zaregistrujte sa: [platform.deepseek.com](https://platform.deepseek.com) -2. Získajte kľúč API -3. Dashboard → Pridať poskytovateľa → DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -**Modely:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +### DeepSeek -1. Zaregistrujte sa: [console.groq.com](https://console.groq.com) -2. Získajte kľúč API (vrátane bezplatnej úrovne) -3. Dashboard → Pridať poskytovateľa → Groq +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -**Modely:**`groq/lama-3.3-70b`, `groq/mixtral-8x7b` +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Tip pre profesionálov:**Ultra rýchle odvodenie – najlepšie pre kódovanie v reálnom čase!### OpenRouter (100+ Models) +### Groq (Free Tier Available!) -1. Zaregistrujte sa: [openrouter.ai](https://openrouter.ai) -2. Získajte kľúč API -3. Dashboard → Pridať poskytovateľa → OpenRouter +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -**Modely:**Získajte prístup k viac ako 100 modelom od všetkých hlavných poskytovateľov prostredníctvom jediného kľúča API. +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Správanie panela:**Modely OpenRouter sú spravované z**Dostupných modelov**. Manuálne pridávanie, import a automatická synchronizácia aktualizujú rovnaký zoznam.
+**Pro Tip:** Ultra-fast inference — best for real-time coding! - -💰 Lacní poskytovatelia (záložní)### GLM-4.7 (Daily reset, $0.6/1M) +### OpenRouter (100+ Models) -1. Zaregistrujte sa: [Zhipu AI](https://open.bigmodel.cn/) -2. Získajte kľúč API z plánu kódovania -3. Dashboard → Pridať kľúč API: - - Poskytovateľ: `glm` - - Kľúč API: „váš kľúč“. +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -**Použitie:**`glm/glm-4.7` +**Models:** Access 100+ models from all major providers through a single API key. -**Tip pre profesionálov:**Kódovací plán ponúka 3× kvótu za 1/7 cenu! Resetovať denne o 10:00.### MiniMax M2.1 (5h reset, $0.20/1M) +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -1. Zaregistrujte sa: [MiniMax](https://www.minimax.io/) -2. Získajte kľúč API -3. Panel → Pridať kľúč API + -**Použitie:**`minimax/MiniMax-M2.1` +
+💰 Cheap Providers (Backup) -**Tip pre profesionálov:**Najlacnejšia možnosť pre dlhý kontext (1 milión tokenov)!### Kimi K2 ($9/month flat) +### GLM-4.7 (Daily reset, $0.6/1M) -1. Prihláste sa na odber: [Moonshot AI](https://platform.moonshot.ai/) -2. Získajte kľúč API -3. Panel → Pridať kľúč API +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**Použite:**„kimi/kimi-najnovšie“. +**Use:** `glm/glm-4.7` -**Tip pre profesionálov:**Pevné 9 $/mesiac za 10 miliónov tokenov = 0,90 $/1 milión efektívnych nákladov!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. - -🆓 BEZPLATNÍ poskytovatelia (núdzové zálohovanie)### Qoder (5 FREE models via OAuth) +### MiniMax M2.1 (5h reset, $0.20/1M) + +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `minimax/MiniMax-M2.1` + +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1563,8 +1787,10 @@ Models:
- -🎨 Vytváranie kombinácií### Example 1: Maximize Subscription → Cheap Backup +
+🎨 Create Combos + +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1592,8 +1818,10 @@ Cost: $0 forever!
- -🔧 Integrácia CLI### Cursor IDE +
+🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1604,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Na konfiguráciu jedným kliknutím použite stránku**CLI Tools**na informačnom paneli alebo upravte súbor `~/.claude/settings.json` manuálne.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1615,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Možnosť 1 – Dashboard (odporúčané):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Možnosť 2 – Manuálne:**Upravte `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1632,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Poznámka:**OpenClaw funguje iba s lokálnou OmniRoute. Použite `127.0.0.1` namiesto `localhost`, aby ste sa vyhli problémom s rozlíšením IPv6.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1646,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Krok 1:**Pridajte OmniRoute ako vlastného poskytovateľa:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Krok 2:**Vytvorte/upravte súbor `opencode.json` v koreňovom adresári projektu:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1672,117 +1909,130 @@ opencode } } } -```` +``` -**Krok 3:**Vyberte model v OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Tip:**Pridajte akýkoľvek model dostupný v koncovom bode vášho OmniRoute `/v1/models` do sekcie `models`. Použite formát `provider/model-id` z hlavného panela OmniRoute.
+ --- ## Riešenie problémov - -Kliknutím rozbalíte sprievodcu riešením problémov +
+Click to expand troubleshooting guide -**„Jazykový model neposkytol správy“** +**"Language model did not provide messages"** -- Kvóta poskytovateľa je vyčerpaná → Skontrolujte sledovanie kvót na paneli -- Riešenie: Použite záložnú kombináciu alebo prejdite na lacnejšiu úroveň +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**Obmedzenie sadzby** +**Rate limiting** -- Vyčerpaná kvóta predplatného → Návrat na GLM/MiniMax -– Pridajte kombináciu: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**Platnosť tokenu OAuth vypršala** +**OAuth token expired** -- Automaticky obnovuje OmniRoute -- Ak problémy pretrvávajú: Dashboard → Provider → Reconnect +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Vysoké náklady** +**High costs** -- Skontrolujte štatistiky používania v hlavnom paneli → Náklady -- Prepnite primárny model na GLM/MiniMax -- Používajte bezplatnú vrstvu (Gemini CLI, Qoder) pre nekritické úlohy +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Porty palubnej dosky/API sú nesprávne** +**Dashboard/API ports are wrong** -- `PORT` je kanonický základný port (predvolene a port API) -- `API_PORT` prepíše iba poslucháč API kompatibilný s OpenAI -- `DASHBOARD_PORT` prepíše iba poslucháč dashboard/Next.js -– Nastavte `NEXT_PUBLIC_BASE_URL` na svoj informačný panel/verejnú adresu URL (pre spätné volania OAuth) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Chyby synchronizácie v cloude** +**Cloud sync errors** -- Overte, že `BASE_URL` odkazuje na vašu spustenú inštanciu -– Overte, že `CLOUD_URL` odkazuje na očakávaný koncový bod cloudu -- Ponechajte hodnoty „NEXT_PUBLIC_*“ zarovnané s hodnotami na strane servera +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Prvé prihlásenie nefunguje** +**First login not working** -– Skontrolujte heslo „INITIAL_PASSWORD“ v súbore „.env“. -- Ak nie je nastavené, záložné heslo je `123456` +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Žiadne záznamy žiadostí** +**No request logs** -- Artefakty požiadavky sa zapisujú do `DATA_DIR/call_logs/` ako jeden súbor JSON na žiadosť -- Povoľte zachytávanie potrubí z ovládacieho panela → Protokoly → Protokoly žiadostí, ak potrebujete podrobné užitočné zaťaženia pre jednotlivé fázy -- Nastavte `APP_LOG_TO_FILE=true`, ak chcete aj protokoly konzoly aplikácie v `logs/application/app.log` -– Podľa potreby upravte hodnoty „APP_LOG_MAX_FILE_SIZE“, „APP_LOG_RETENTION_DAYS“, „APP_LOG_MAX_FILES“ a „CALL_LOG_MAX_ENTRIES“ +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Test pripojenia ukazuje „Neplatné“ pre poskytovateľov kompatibilných s OpenAI** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Mnoho poskytovateľov nezverejňuje koncový bod `/models` -- OmniRoute v1.0.6+ zahŕňa záložné overenie prostredníctvom dokončenia chatu -- Uistite sa, že základná adresa URL obsahuje príponu `/v1`### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ Dôležité pre používateľov, ktorí používajú OmniRoute na VPS, Docker alebo akomkoľvek vzdialenom serveri**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -Poskytovatelia**Antigravity**a**Gemini CLI**používajú**Google OAuth 2.0**. Google vyžaduje, aby sa parameter „redirect_uri“ v toku OAuth presne zhodoval s jedným z vopred zaregistrovaných identifikátorov URI v konzole Google Cloud Console aplikácie. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -Prihlasovacie údaje OAuth v balíku v OmniRoute sú registrované**iba pre `localhost`**. Keď pristupujete k OmniRoute na vzdialenom serveri (napr. `https://omniroute.myserver.com`), Google odmietne overenie pomocou:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -V službe Google Cloud Console musíte vytvoriť**OAuth 2.0 Client ID**s identifikátorom URI vášho servera.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Otvoriť Google Cloud Console** +#### Step-by-step -Prejdite na: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Vytvorte nové ID klienta OAuth 2.0** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Kliknite na**"+ Vytvoriť poverenia"**→**"ID klienta OAuth"** -- Typ aplikácie:**"Webová aplikácia"** -- Názov: čokoľvek, čo sa vám páči (napr. `OmniRoute Remote`) +**2. Create a new OAuth 2.0 Client ID** -**3. Pridať identifikátory URI autorizovaného presmerovania** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -Do poľa**"URI autorizovaného presmerovania"**pridajte:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Nahraďte `vas-server.com` doménou alebo IP svojho servera (v prípade potreby uveďte port, napr. `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. Uložte a skopírujte poverenia** +After creating, Google will show the **Client ID** and **Client Secret**. -Po vytvorení Google zobrazí**Client ID**a**Client Secret**. +**5. Set environment variables** -**5. Nastaviť premenné prostredia** +In your `.env` (or Docker environment variables): -Vo vašom `.env` (alebo premenných prostredia Docker):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1791,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Reštartujte OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Skúste sa pripojiť znova** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Dashboard → Poskytovatelia → Antigravitácia (alebo Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google sa teraz správne presmeruje na `https://your-server.com/callback`.--- +--- #### Temporary workaround (without custom credentials) -Ak si práve teraz nechcete nastaviť svoje vlastné poverenia, stále môžete použiť**manuálny postup webovej adresy**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute otvorí autorizačnú URL Google -2. Po autorizácii sa Google pokúsi presmerovať na `localhost` (čo zlyhá na vzdialenom serveri) -3.**Skopírujte celú adresu URL**z panela s adresou prehliadača (aj keď sa stránka nenačíta) -4. Prilepte túto adresu URL do poľa zobrazeného v režime pripojenia OmniRoute -5. Kliknite na**"Pripojiť"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Funguje to, pretože autorizačný kód v adrese URL je platný bez ohľadu na to, či sa stránka s presmerovaním načítala.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. - -🇧🇷 Versão em Português#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Osvedčuje**Antigravity**a**Gemini CLI**používame**Google OAuth 2.0**ako autentifikáciu. O Google exige que a `redirect_uri` sa používa bez toku OAuth saja**exatamente**uma das URI pre-kadastradas no Google Cloud Console sa nedá použiť. +
+🇧🇷 Versão em Português -Ako credenciais OAuth embutidas no OmniRoute estão cadastradas**apenas para `localhost`**. K dispozícii je prístup k OmniRoute k vzdialenému servisu (napr.: „https://omniroute.meuservidor.com“), alebo k autenticite spoločnosti Google:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Presné údaje sú**OAuth 2.0 Client ID**bez služby Google Cloud Console s identifikátorom URI.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Prístup k službe Google Cloud Console** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) **2. Crie um novo OAuth 2.0 Client ID** -- Kliknite na**"+ Vytvoriť poverenia"**→**"ID klienta OAuth"** -- Tipo de aplicativo:**"Webová aplikácia"** -- Nome: escolha qualquer nome (napr. `OmniRoute Remote`) +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Adicione ako Authorized Redirect URI** +**3. Adicione as Authorized Redirect URIs** -Žiadne pole**"URI autorizovaného presmerovania"**, adicione:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, napr.: `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. Uložiť a kópiu ako poverenie** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Após criar, o Google mostrará o**Client ID**e o**Client Secret**. +**5. Configure as variáveis de ambiente** -**5. Konfigurovať ako variáveis de ambiente** +No seu `.env` (ou nas variáveis de ambiente do Docker): -No seu `.env` (ou nas variáveis de ambiente do Docker):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1870,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reinicie alebo OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute - -```` +``` **7. Tente conectar novamente** -Dashboard → Poskytovatelia → Antigravitácia (alebo Gemini CLI) → OAuth +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Agora alebo Google presmerovaný na adresu `https://seu-servidor.com/callback` a autentickú funkciu.--- +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. + +--- #### Workaround temporário (sem configurar credenciais próprias) -Ak chcete získať prístup k dôvere, môžete použiť**príručku URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. O OmniRoute abrirá a URL autorização Google +1. O OmniRoute abrirá a URL de autorização do Google 2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) -3.**Skopírujte úplnú webovú adresu**da barra de endereço do seu browser (mesmo que a pagina não carregue) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) 4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute -5. Kliknite na**"Pripojiť"** +5. Clique em **"Connect"** -> Toto riešenie funguje pomocou autorizačného kódu na adrese URL a nezávislého presmerovania.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1908,64 +2171,73 @@ Ak chcete získať prístup k dôvere, môžete použiť**príručku URL**: ## 🛠️ Tech Stack - -Kliknutím rozbalíte podrobnosti o technickom zásobníku +
+Click to expand tech stack details --**Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ nie je**podporovaný**– natívne binárne súbory `better-sqlite3` sú nekompatibilné) --**Jazyk**: TypeScript 5.9 —**100 % TypeScript**cez `src/` a `open-sse/` (nula `any` v základných moduloch od verzie 2.0) -–**Framework**: Next.js 16 + React 19 + Tailwind CSS 4 --**Databáza**: LowDB (JSON) + SQLite (stav domény + protokoly proxy + audit MCP + rozhodnutia o smerovaní) --**Schémy**: Zod (overenie I/O nástroja MCP, zmluvy API) --**Protokoly**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Streamovanie**: Server-Sent Events (SSE) --**Auth**: OAuth 2.0 (PKCE) + JWT + kľúče API + autorizácia v rozsahu MCP --**Testovanie**: Testovací program Node.js + Vitest (900+ testov vrátane jednotky, integrácie, E2E) --**CI/CD**: Akcie GitHub (automatické zverejňovanie npm + Docker Hub pri vydaní) --**Web**: [omniroute.online](https://omniroute.online) -–**Balík**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) -–**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Odolnosť**: Istič, exponenciálne ustupovanie, stádo proti hromu, spoofing TLS, auto-kombinované samoliečenie
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Dokumentácia -| Dokument | Popis | -| ----------------------------------------------- | ---------------------------------------------------- | -| [Používateľská príručka](docs/USER_GUIDE.md) | Poskytovatelia, kombá, integrácia CLI, nasadenie | -| [Odkaz na API](docs/API_REFERENCE.md) | Všetky koncové body s príkladmi | -| [Server MCP](open-sse/mcp-server/README.md) | 16 nástrojov MCP, konfigurácií IDE, klientov Python/TS/Go | -| [Server A2A](src/lib/a2a/README.md) | Protokol JSON-RPC 2.0, zručnosti, streamovanie, správa úloh | -| [Auto-Combo Engine](docs/auto-combo.md) | 6-faktorové bodovanie, balíčky režimov, samoliečba | -| [Riešenie problémov](docs/TROUBLESHOOTING.md) | Bežné problémy a riešenia | -| [Architektúra](docs/ARCHITECTURE.md) | Architektúra systému a vnútorné vybavenie | -| [Prispievanie](CONTRIBUTING.md) | Nastavenie vývoja a usmernenia | -| [Špecifikácia OpenAPI](docs/openapi.yaml) | Špecifikácia OpenAPI 3.0 | -| [Bezpečnostná politika](SECURITY.md) | Nahlasovanie zraniteľnosti a bezpečnostné postupy | -| [Nasadenie VM](docs/VM_DEPLOYMENT_GUIDE.md) | Kompletný sprievodca: nastavenie VM + nginx + Cloudflare | -| [Galéria funkcií](docs/FEATURES.md) | Vizuálna prehliadka prístrojového panela so snímkami obrazovky | -| [Kontrolný zoznam vydania](docs/RELEASE_CHECKLIST.md) | Kroky overenia pred vydaním |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute má naplánovaných**210+ funkcií**vo viacerých fázach vývoja. Tu sú kľúčové oblasti: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Kategória | Plánované funkcie | Najdôležitejšie | -| ------------------------------ | ----------------- | -------------------------------------------------------------------------------------- | -| 🧠**Smerovanie a inteligencia**| 25+ | Smerovanie s najnižšou latenciou, smerovanie založené na značkách, predbežná kontrola kvóty, výber účtu P2C | -| 🔒**Bezpečnosť a dodržiavanie predpisov**| 20+ | Spevnenie SSRF, maskovanie poverení, limit rýchlosti na koncový bod, rozsah kľúča riadenia | -| 📊**Pozorovateľnosť**| 15+ | Integrácia OpenTelemetry, monitorovanie kvót v reálnom čase, sledovanie nákladov na model | -| 🔄**Integrácie poskytovateľov**| 20+ | Register dynamických modelov, cooldowny poskytovateľov, kódex pre viacero účtov, analýza kvót Copilota | -| ⚡**Výkon**| 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | -| 🌐**Ekosystém**| 10+ | WebSocket API, rýchle opätovné načítanie konfigurácie, distribuovaný ukladací priestor konfigurácií, komerčný režim |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**Integrácia OpenCode**– podpora natívneho poskytovateľa pre IDE kódovania OpenCode AI -- 🔗**Integrácia TRAE**– Úplná podpora pre vývojový rámec TRAE AI -- 📦**Batch API**– Asynchrónne dávkové spracovanie pre hromadné požiadavky -- 🎯**Smerovanie založené na značkách**– Smerujte požiadavky na základe vlastných značiek a metadát -- 💰**Stratégia najnižšej ceny**— Automaticky vyberte najlacnejšieho dostupného poskytovateľa +### 🔜 Coming Soon -> 📝 Úplné špecifikácie funkcií dostupné v [`docs/new-features/`](docs/new-features/) (217 podrobných špecifikácií)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1973,18 +2245,20 @@ OmniRoute má naplánovaných**210+ funkcií**vo viacerých fázach vývoja. Tu ### How to Contribute -1. Rozdeľte úložisko -2. Vytvorte si vetvu funkcií (`git checkout -b feature/amazing-feature`) -3. Potvrdiť zmeny (`git commit -m 'Pridať úžasnú funkciu'`) -4. Push to branch (`git push origin feature/amazing-feature`) -5. Otvorte požiadavku na stiahnutie +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Podrobné pokyny nájdete na stránke [CONTRIBUTING.md](CONTRIBUTING.md).### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1996,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Špeciálne poďakovanie patrí**[9router](https://github.com/decolua/9router)**od**[decolua](https://github.com/decolua)**– pôvodnému projektu, ktorý inšpiroval tento fork. OmniRoute stavia na tomto neuveriteľnom základe s ďalšími funkciami, multimodálnymi API a úplným prepísaním TypeScript. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Špeciálne poďakovanie patrí**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**– pôvodnej implementácii Go, ktorá inšpirovala tento port JavaScript.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Licencia -Licencia MIT – podrobnosti nájdete v [LICENCIA](LICENCIA).--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/sk/docs/ARCHITECTURE.md b/docs/i18n/sk/docs/ARCHITECTURE.md index 5f0b3ede48..0447587be4 100644 --- a/docs/i18n/sk/docs/ARCHITECTURE.md +++ b/docs/i18n/sk/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Posledná aktualizácia: 28.03.2026_## Executive Summary -OmniRoute je lokálna AI smerovacia brána a dashboard postavená na Next.js. -Poskytuje jeden koncový bod kompatibilný s OpenAI (`/v1/*`) a nasmeruje prevádzku naprieč viacerými poskytovateľmi upstream s prekladom, záložným, obnovovaním tokenov a sledovaním používania. -Základné schopnosti: +_Last updated: 2026-03-28_ -- OpenAI kompatibilný povrch API pre CLI/nástroje (28 poskytovateľov) -- Požiadavka / odpoveď na preklad medzi formátmi poskytovateľov -- Záložná kombinácia modelov (sekvencia viacerých modelov) - – Záložný režim na úrovni účtu (viac účtov na poskytovateľa) -- Správa pripojenia poskytovateľa s kľúčom OAuth + API -- Generovanie vkladania cez `/v1/embeddings` (6 poskytovateľov, 9 modelov) -- Generovanie obrázkov prostredníctvom `/v1/images/generations` (4 poskytovatelia, 9 modelov) -- Myslite na analýzu značiek (`...`) pre modely uvažovania -- Dezinfekcia odozvy pre prísnu kompatibilitu OpenAI SDK -- Normalizácia rolí (vývojár→systém, systém→používateľ) pre kompatibilitu medzi poskytovateľmi -- Konverzia štruktúrovaného výstupu (json_schema → Gemini responseSchema) -- Miestna perzistencia pre poskytovateľov, kľúče, aliasy, kombá, nastavenia, ceny -- Sledovanie používania / nákladov a zaznamenávanie žiadostí -- Voliteľná cloudová synchronizácia pre synchronizáciu viacerých zariadení/stavov -- Zoznam povolených/blokovaných IP adries pre riadenie prístupu k API -- Myslenie na správu rozpočtu (priechodový/automatický/vlastný/adaptívny) -- Rýchle vstrekovanie globálneho systému -- Sledovanie relácií a snímanie odtlačkov prstov -- Rozšírené obmedzenie sadzieb na účet s profilmi špecifickými pre poskytovateľov -- Vzor ističa pre odolnosť poskytovateľa -- Ochrana stáda proti hromu s blokovaním mutex -- Cache deduplikácie požiadaviek na základe podpisu -- Doménová vrstva: dostupnosť modelu, cenové pravidlá, záložná politika, politika blokovania -- Stálosť stavu domény (vyrovnávacia pamäť SQLite pre záložné zdroje, rozpočty, blokovania, ističe) -- Modul politiky pre centralizované vyhodnocovanie požiadaviek (uzamknutie → rozpočet → záložné) -- Požiadajte o telemetriu s agregáciou latencie p50/p95/p99 -- ID korelácie (X-Request-Id) pre end-to-end sledovanie -- Protokolovanie auditu súladu s odhlásením podľa kľúča API -- Hodnotný rámec pre zabezpečenie kvality LLM -- Prístrojová doska UI Resilience so stavom ističa v reálnom čase -- Modulárni poskytovatelia OAuth (12 samostatných modulov pod `src/lib/oauth/providers/`) +## Executive Summary -Primárny runtime model: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- Trasy aplikácií Next.js pod `src/app/api/*` implementujú rozhrania API dashboardu aj rozhrania API kompatibility -- Zdieľané jadro SSE/smerovanie v `src/sse/*` + `open-sse/*` sa stará o vykonávanie poskytovateľa, preklad, streamovanie, záložné zdroje a používanie## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Runtime lokálnej brány -- Rozhrania API na správu informačných panelov -- Overenie poskytovateľa a obnovenie tokenu -- Požiadajte o preklad a streamovanie SSE -- Miestny stav + pretrvávanie používania -- Voliteľná orchestrácia synchronizácie s cloudom### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Implementácia cloudovej služby za `NEXT_PUBLIC_CLOUD_URL` -- Poskytovateľ SLA/riadiaca rovina mimo lokálneho procesu -- Samotné externé binárne súbory CLI (Claude CLI, Codex CLI atď.)## Dashboard Surface (Current) +### Out of Scope -Hlavné stránky pod `src/app/(dashboard)/dashboard/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — rýchly štart + prehľad poskytovateľa -- `/dashboard/endpoint` — proxy koncového bodu + MCP + A2A + karty koncového bodu rozhrania API -- `/dashboard/providers` – pripojenia a poverenia poskytovateľa -- `/dashboard/combos` — kombinované stratégie, šablóny, pravidlá smerovania modelov -- `/dashboard/costs` – agregácia nákladov a viditeľnosť cien -- `/dashboard/analytics` – analýzy a hodnotenia používania -- „/dashboard/limits“ – kontroly kvót/sadzieb -- `/dashboard/cli-tools` — CLI onboarding, runtime detekcia, generovanie konfigurácie -- `/dashboard/agents` — zistení agenti AKT + registrácia vlastného agenta -- `/dashboard/media` – ihrisko s obrázkami/videom/hudbou -- `/dashboard/search-tools` — testovanie a história poskytovateľa vyhľadávania -- „/dashboard/health“ – doba prevádzkyschopnosti, ističe, limity sadzieb -- `/dashboard/logs` — denníky požiadaviek/proxy/audit/konzoly -- `/dashboard/settings` — karty systémových nastavení (všeobecné, smerovanie, predvolené nastavenia komba atď.) -- `/dashboard/api-manager` — životný cyklus kľúča API a povolenia modelu## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Hlavné adresáre: +Main directories: -- `src/app/api/v1/*` a `src/app/api/v1beta/*` pre rozhrania API kompatibility -- `src/app/api/*` pre spravovanie/konfiguráciu API -- Ďalej prepíše mapu `next.config.mjs` `/v1/*` na `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Dôležité cesty kompatibility: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` — zahŕňa vlastné modely s `custom: true` -- `src/app/api/v1/embeddings/route.ts` — generovanie vloženia (6 poskytovateľov) -- `src/app/api/v1/images/generations/route.ts` — generovanie obrázkov (4+ poskytovatelia vrátane Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` - – `src/app/api/v1/providers/[poskytovateľ]/chat/completions/route.ts` – vyhradený chat pre jednotlivých poskytovateľov - – `src/app/api/v1/providers/[poskytovateľ]/embeddings/route.ts` – vyhradené vloženia podľa jednotlivých poskytovateľov - – `src/app/api/v1/providers/[poskytovateľ]/images/generations/route.ts` – vyhradené obrázky podľa jednotlivých poskytovateľov +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` -- `src/app/api/v1beta/models/[...cesta]/route.ts` +- `src/app/api/v1beta/models/[...path]/route.ts` -Manažérske domény: +Management domains: - Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` -- Poskytovatelia/pripojenia: `src/app/api/providers*` -- Uzly poskytovateľa: `src/app/api/provider-nodes*` -- Vlastné modely: `src/app/api/provider-models` (GET/POST/DELETE) -- Katalóg modelov: `src/app/api/models/route.ts` (GET) - – Konfigurácia proxy: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` - – Kľúče/aliasy/kombá/cena: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- Použitie: `src/app/api/usage/*` +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` - Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` -- Pomocníci nástrojov CLI: `src/app/api/cli-tools/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` - IP filter: `src/app/api/settings/ip-filter` (GET/PUT) - Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) -- Systémová výzva: `src/app/api/settings/system-prompt` (GET/PUT) -- Relácie: `src/app/api/sessions` (GET) -- Limity sadzby: `src/app/api/rate-limits` (GET) -- Odolnosť: `src/app/api/resilience` (GET/PATCH) – profily poskytovateľa, istič, stav limitu rýchlosti -- Resetovanie odolnosti: `src/app/api/resilience/reset` (POST) — reset ističov + cooldowny -- Štatistiky vyrovnávacej pamäte: `src/app/api/cache/stats` (GET/DELETE) -- Dostupnosť modelu: `src/app/api/models/availability` (GET/POST) - – Telemetria: `src/app/api/telemetry/summary` (GET) - – Rozpočet: `src/app/api/usage/budget` (GET/POST) -- Záložné reťazce: `src/app/api/fallback/chains` (GET/POST/DELETE) -- Audit súladu: `src/app/api/compliance/audit-log` (GET) -- Hodnoty: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- Zásady: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -Hlavné prietokové moduly: +## 2) SSE + Translation Core -- Záznam: `src/sse/handlers/chat.ts` -- Základná orchestrácia: `open-sse/handlers/chatCore.ts` -- Spúšťacie adaptéry poskytovateľa: `open-sse/executors/*` -- Detekcia formátu/konfigurácia poskytovateľa: `open-sse/services/provider.ts` -- Analýza/rozlíšenie modelu: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Logika záložného účtu: `open-sse/services/accountFallback.ts` -- Register prekladov: `open-sse/translator/index.ts` -- Transformácie streamu: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- Extrakcia/normalizácia použitia: `open-sse/utils/usageTracking.ts` -- Analyzátor značiek Think: `open-sse/utils/thinkTagParser.ts` -- Ovládač vkladania: `open-sse/handlers/embeddings.ts` -- Register poskytovateľa vkladania: `open-sse/config/embeddingRegistry.ts` -- Obslužný program generovania obrázkov: `open-sse/handlers/imageGeneration.ts` -- Register poskytovateľa obrázkov: `open-sse/config/imageRegistry.ts` -- Dezinfekcia odozvy: `open-sse/handlers/responseSanitizer.ts` -- Normalizácia rolí: `open-sse/services/roleNormalizer.ts` +Main flow modules: -Služby (obchodná logika): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- Výber účtu/bodovanie: `open-sse/services/accountSelector.ts` -- Kontextová správa životného cyklu: `open-sse/services/contextManager.ts` -- Vynútenie filtra IP: `open-sse/services/ipFilter.ts` -- Sledovanie relácií: `open-sse/services/sessionManager.ts` -- Žiadosť o deduplikáciu: `open-sse/services/signatureCache.ts` -- Vloženie systémovej výzvy: `open-sse/services/systemPrompt.ts` -- Myslenie na riadenie rozpočtu: `open-sse/services/thinkingBudget.ts` -- Smerovanie modelu so zástupnými znakmi: `open-sse/services/wildcardRouter.ts` -- Správa limitov sadzieb: `open-sse/services/rateLimitManager.ts` -- Istič: `open-sse/services/circuitBreaker.ts` +Services (business logic): -Moduly vrstvy domény: +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- Dostupnosť modelu: `src/lib/domain/modelAvailability.ts` -- Cenové pravidlá/rozpočty: `src/lib/domain/costRules.ts` -- Záložná politika: `src/lib/domain/fallbackPolicy.ts` -- Kombinovaný prekladač: `src/lib/domain/comboResolver.ts` -- Zásady blokovania: `src/lib/domain/lockoutPolicy.ts` -- Modul politiky: `src/domain/policyEngine.ts` – centralizované uzamknutie → rozpočet → záložné hodnotenie -- Katalóg kódov chýb: `src/lib/domain/errorCodes.ts` - – ID požiadavky: `src/lib/domain/requestId.ts` -- Časový limit načítania: `src/lib/domain/fetchTimeout.ts` -- Požiadať o telemetriu: `src/lib/domain/requestTelemetry.ts` -- Súlad/audit: `src/lib/domain/compliance/index.ts` +Domain layer modules: + +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` - Eval runner: `src/lib/domain/evalRunner.ts` -- Trvalosť stavu domény: `src/lib/db/domainState.ts` — SQLite CRUD pre záložné reťazce, rozpočty, históriu nákladov, stav uzamknutia, ističe +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -Moduly poskytovateľa OAuth (12 samostatných súborov pod `src/lib/oauth/providers/`): +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -- Index registra: `src/lib/oauth/providers/index.ts` - – Jednotliví poskytovatelia: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilo`coded, `kilo`` -- Tenký obal: `src/lib/oauth/providers.ts` — reexporty z jednotlivých modulov## 3) Persistence Layer +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -Primárny stav DB (SQLite): +## 3) Persistence Layer -- Infračervené jadro: `src/lib/db/core.ts` (better-sqlite3, migrácie, WAL) -- Fasáda opätovného exportu: `src/lib/localDb.ts` (tenká vrstva kompatibility pre volajúcich) -- súbor: `${DATA_DIR}/storage.sqlite` (alebo `$XDG_CONFIG_HOME/omniroute/storage.sqlite`, ak je nastavený, inak `~/.omniroute/storage.sqlite`) -- entity (tabuľky + priestory názvov KV): providerConnections, providerNodes, modelAliases, kombá, apiKeys, nastavenia, ceny,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +Primary state DB (SQLite): -Trvanlivosť pri používaní: +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -- fasáda: `src/lib/usageDb.ts` (rozložené moduly v `src/lib/usage/*`) -- SQLite tabuľky v `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` -- voliteľné artefakty súboru zostávajú kvôli kompatibilite/ladeniu (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- staršie súbory JSON sa migrujú na SQLite migráciami pri spustení, ak sú k dispozícii +Usage persistence: -DB stavu domény (SQLite): +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- `src/lib/db/domainState.ts` — operácie CRUD pre stav domény - – Tabuľky (vytvorené v `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `doména_cost_history`, `domain_lockout_state`, `prerušovače_obvodu_domény` -- Vzor vyrovnávacej pamäte pre zápis: mapy v pamäti sú autoritatívne za behu; mutácie sa zapisujú synchrónne do SQLite; stav sa obnoví z DB pri studenom štarte## 4) Auth + Security Surfaces +Domain State DB (SQLite): -- Overenie súboru cookie informačného panela: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- Generovanie/overenie kľúča API: `src/shared/utils/apiKey.ts` -- Tajomstvá poskytovateľa pretrvávali v záznamoch `providerConnections` -- Podpora odchádzajúceho servera proxy cez `open-sse/utils/proxyFetch.ts` (env vars) a `open-sse/utils/networkProxy.ts` (konfigurovateľné podľa jednotlivých poskytovateľov alebo globálne)## 5) Cloud Sync +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start -- Spustenie plánovača: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Pravidelná úloha: `src/shared/services/cloudSyncScheduler.ts` -- Pravidelná úloha: `src/shared/services/modelSyncScheduler.ts` -- Riadiaca cesta: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Záložné rozhodnutia riadi `open-sse/services/accountFallback.ts` pomocou stavových kódov a heuristiky chybových správ. Kombinované smerovanie pridáva ďalšiu ochranu: 400 v rozsahu poskytovateľa, ako sú zlyhania blokovania obsahu a overovania rolí, sa považujú za lokálne zlyhania modelu, takže neskoršie kombinované ciele môžu stále bežať.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -Obnovenie počas živej prevádzky sa vykonáva v rámci `open-sse/handlers/chatCore.ts` prostredníctvom spúšťača `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -Pravidelnú synchronizáciu spúšťa „CloudSyncScheduler“, keď je povolený cloud.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -Súbory fyzického úložiska: +Physical storage files: -- primárny runtime DB: `${DATA_DIR}/storage.sqlite` -- riadky denníka žiadostí: `${DATA_DIR}/log.txt` (artefakt compat/debug) -- archívy štruktúrovaného obsahu hovorov: `${DATA_DIR}/call_logs/` -- voliteľné relácie ladenia prekladača/požiadavky: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: rozhrania API pre kompatibilitu -- `src/app/api/v1/providers/[poskytovateľ]/*`: vyhradené trasy pre jednotlivých poskytovateľov (chat, vkladanie, obrázky) -- `src/app/api/providers*`: poskytovateľ CRUD, validácia, testovanie -- `src/app/api/provider-nodes*`: vlastná správa kompatibilných uzlov -- `src/app/api/provider-models`: správa vlastných modelov (CRUD) -- `src/app/api/models/route.ts`: API katalógu modelov (aliasy + vlastné modely) -- `src/app/api/oauth/*`: toky OAuth/kódu zariadenia -- `src/app/api/keys*`: životný cyklus lokálneho kľúča API -- `src/app/api/models/alias`: správa aliasov -- `src/app/api/combos*`: správa náhradných kombinácií -- `src/app/api/pricing`: prepíše ceny pre výpočet nákladov -- `src/app/api/settings/proxy`: konfigurácia proxy (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: test externej konektivity (POST) -- `src/app/api/usage/*`: využitie a protokoly API -- `src/app/api/sync/*` + `src/app/api/cloud/*`: synchronizácia v cloude a pomocníci pre cloud -- `src/app/api/cli-tools/*`: miestne zapisovače/kontroly konfigurácie CLI -- `src/app/api/settings/ip-filter`: zoznam povolených adries IP/zoznam blokovaných adries (GET/PUT) -- `src/app/api/settings/thinking-budget`: konfigurácia rozpočtu tokenu myslenia (GET/PUT) -- `src/app/api/settings/system-prompt`: globálna systémová výzva (GET/PUT) -- `src/app/api/sessions`: zoznam aktívnych relácií (GET) -- `src/app/api/rate-limits`: stav limitu sadzby na účet (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: analýza požiadaviek, spracovanie komb, slučka výberu účtu -- `open-sse/handlers/chatCore.ts`: preklad, odoslanie vykonávateľa, spracovanie opakovania/obnovenia, nastavenie streamu -- `open-sse/executors/*`: správanie siete a formátu špecifické pre poskytovateľa### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: register a orchestrácia prekladateľov -- Žiadosť o prekladateľov: `open-sse/translator/request/*` -- Prekladače odpovedí: `open-sse/translator/response/*` -- Formátové konštanty: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: trvalá konfigurácia/stav a trvalosť domény na SQLite -- `src/lib/localDb.ts`: reexport kompatibility pre DB moduly -- `src/lib/usageDb.ts`: fasáda histórie používania/protokolov hovorov nad tabuľkami SQLite## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Každý poskytovateľ má špecializovaný spúšťač rozširujúci `BaseExecutor` (v `open-sse/executors/base.ts`), ktorý poskytuje vytváranie adries URL, konštrukciu hlavičiek, opakovanie s exponenciálnym stiahnutím, háčiky na obnovenie poverení a metódu orchestrácie `execute()`. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Exekútor | Poskytovatelia | Špeciálna manipulácia | -| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------ | -| "DefaultExecutor" | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Konfigurácia dynamickej adresy URL/hlavičky podľa poskytovateľa | -| "AntigravityExecutor" | Google Antigravity | Vlastné ID projektu/relácie, Opakovať po analýze | -| "CodexExecutor" | Kódex OpenAI | Vkladá pokyny systému, vynucuje úsilie na uvažovanie | -| "CursorExecutor" | Kurzor IDE | Protokol ConnectRPC, kódovanie Protobuf, podpis požiadavky cez kontrolný súčet | -| "GithubExecutor" | GitHub Copilot | Obnovenie tokenu kopilota, hlavičky napodobňujúce VSCode | -| "KiroExecutor" | AWS CodeWhisperer/Kiro | Binárny formát AWS EventStream → Konverzia SSE | -| "GeminiCLIExecutor" | Gemini CLI | Cyklus obnovenia tokenu Google OAuth | +### Persistence -Všetci ostatní poskytovatelia (vrátane vlastných kompatibilných uzlov) používajú `DefaultExecutor`.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Poskytovateľ | Formát | Auth | Stream | Nestreamovať | Obnovenie tokenu | Použitie API | -| ---------------- | ---------------- | ----------------------- | -------------------- | ------------ | ---------------- | ------------------- | ------------------------------ | -| Claude | claude | Kľúč API / OAuth | ✅ | ✅ | ✅ | ⚠️ Len správca | -| Blíženci | Blíženci | Kľúč API / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloudová konzola | -| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloudová konzola | -| Antigravitácia | antigravitácia | OAuth | ✅ | ✅ | ✅ | ✅ Plná kvóta API | -| OpenAI | openai | API kľúč | ✅ | ✅ | ❌ | ❌ | -| Kódex | openai-responses | OAuth | ✅ nútený | ❌ | ✅ | ✅ Sadzobné limity | -| GitHub Copilot | openai | OAuth + token Copilot | ✅ | ✅ | ✅ | ✅ Snímky kvóty | -| Kurzor | kurzor | Vlastný kontrolný súčet | ✅ | ✅ | ❌ | ❌ | -| Kiro | kiro | AWS SSO OIDC | ✅ (Stream udalostí) | ❌ | ✅ | ✅ Limity použitia | -| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Na požiadanie | -| Qoder | openai | OAuth (základné) | ✅ | ✅ | ✅ | ⚠️ Na požiadanie | -| OpenRouter | openai | API kľúč | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | claude | API kľúč | ✅ | ✅ | ❌ | ❌ | -| DeepSeek | openai | API kľúč | ✅ | ✅ | ❌ | ❌ | -| Groq | openai | API kľúč | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | openai | API kľúč | ✅ | ✅ | ❌ | ❌ | -| Mistral | openai | API kľúč | ✅ | ✅ | ❌ | ❌ | -| Zmätok | openai | API kľúč | ✅ | ✅ | ❌ | ❌ | -| Spolu AI | openai | API kľúč | ✅ | ✅ | ❌ | ❌ | -| Ohňostroje AI | openai | API kľúč | ✅ | ✅ | ❌ | ❌ | -| Cerebras | openai | API kľúč | ✅ | ✅ | ❌ | ❌ | -| Cohere | openai | API kľúč | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | openai | API kľúč | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Medzi zistené zdrojové formáty patria: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. -- "openai". -- "openai-odpovede". -- "claude". -- "blíženec". +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | -Cieľové formáty zahŕňajú: +All other providers (including custom compatible nodes) use the `DefaultExecutor`. -- OpenAI chat/reakcie +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: + +- `openai` +- `openai-responses` +- `claude` +- `gemini` + +Target formats include: + +- OpenAI chat/Responses - Claude -- Gemini/Gemini-CLI/Antigravitačná obálka +- Gemini/Gemini-CLI/Antigravity envelope - Kiro -- Kurzor +- Cursor -Preklady používajú**OpenAI ako formát centra**— všetky konverzie prechádzajú cez OpenAI ako medziprodukt:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Preklady sa vyberajú dynamicky na základe tvaru zdroja a cieľového formátu poskytovateľa. +Additional processing layers in the translation pipeline: -Ďalšie vrstvy spracovania v reťazci prekladu: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` -–**Dezinfekcia odpovedí**– Odstráni neštandardné polia z odpovedí vo formáte OpenAI (streamovaných aj nestreamovaných), aby sa zabezpečil prísny súlad so súpravou SDK --**Normalizácia rolí**– Konvertuje „vývojár“ → „systém“ pre ciele, ktoré nie sú OpenAI; zlučuje „systém“ → „používateľ“ pre modely, ktoré odmietajú systémovú rolu (GLM, ERNIE) -–**Think tagextrakcia**– analyzuje bloky „...“ z obsahu do poľa „reasoning_content“ -–**Štruktúrovaný výstup**– Konvertuje OpenAI `response_format.json_schema` na Gemini `responseMimeType` + `responseSchema`## Supported API Endpoints +## Supported API Endpoints -| Koncový bod | Formát | Psovod | -| --------------------------------------------------- | ------------------- | -------------------------------------------------------------------- | -| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | -| `POST /v1/messages` | Claude Správy | Rovnaký handler (automaticky detekovaný) | -| `POST /v1/responses` | Odpovede OpenAI | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | -| "GET /v1/embeddings" | Zoznam modelov | Cesta API | -| `POST /v1/images/generations` | Obrázky OpenAI | `open-sse/handlers/imageGeneration.ts` | -| "GET /v1/images/generations" | Zoznam modelov | Cesta API | -| `POST /v1/providers/{poskytovateľ}/chat/completions` | OpenAI Chat | Vyhradené pre každého poskytovateľa s overením modelu | -| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Vyhradené pre každého poskytovateľa s overením modelu | -| `POST /v1/providers/{poskytovateľ}/images/generations` | Obrázky OpenAI | Vyhradené pre každého poskytovateľa s overením modelu | -| `POST /v1/messages/count_tokens` | Počet tokenov Claude | Cesta API | -| "GET /v1/models" | Zoznam modelov OpenAI | Cesta API (chat + vkladanie + obrázok + vlastné modely) | -| "GET /api/models/catalog" | Katalóg | Všetky modely zoskupené podľa poskytovateľa + typ | -| `POST /v1beta/models/*:streamGenerateContent` | Rodák Blíženci | Cesta API | -| "GET/PUT/DELETE /api/settings/proxy" | Konfigurácia proxy | Konfigurácia sieťového proxy | -| `POST /api/settings/proxy/test` | Pripojenie proxy | Koncový bod testu stavu proxy/konektivity | -| "ZÍSKAŤ/POST/DELETE /api/modely poskytovateľa" | Modely poskytovateľov | Vlastné a spravované dostupné modely podporujúce metadáta modelu poskytovateľa |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -Obídený obslužný program (`open-sse/utils/bypassHandler.ts`) zachytí známe požiadavky na „zahodenie“ z Claude CLI – zahrievacie pingy, extrakcie titulov a počty tokenov – a vráti**falošnú odpoveď**bez spotrebovania tokenov poskytovateľa upstream. Toto sa spustí iba vtedy, keď „User-Agent“ obsahuje „claude-cli“.## Request Logger Pipeline +## Bypass Handler -Záznamník požiadaviek (`open-sse/utils/requestLogger.ts`) poskytuje 7-stupňový kanál na zaznamenávanie ladenia, ktorý je predvolene vypnutý, povolený cez `ENABLE_REQUEST_LOGS=true`:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Súbory sa zapisujú do `/logs//` pre každú reláciu požiadavky.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- Ochladenie účtu poskytovateľa pri prechodných chybách/chybách rýchlosti/autorizácie -- záložný účet pred neúspešnou žiadosťou -- záložný kombinovaný model, keď je vyčerpaná aktuálna cesta modelu/poskytovateľa## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- predbežná kontrola a obnovenie s opätovným pokusom pre poskytovateľov obnoviteľných zdrojov -- 401/403 zopakovanie po pokuse o obnovenie v základnej ceste## 3) Stream Safety +## 2) Token Expiry -- regulátor prúdu s vedomím odpojenia -- prekladový tok s vyprázdnením konca toku a spracovaním `[DONE]` -- záložný odhad použitia, keď chýbajú metadáta používania poskytovateľa## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- Objavia sa chyby synchronizácie, ale lokálny runtime pokračuje -- plánovač má logiku schopnú opakovania, ale pravidelné vykonávanie v súčasnosti štandardne volá synchronizáciu na jeden pokus## 5) Data Integrity +## 3) Stream Safety -- Migrácie schém SQLite a automatické aktualizácie pri spustení -- staršia cesta kompatibility migrácie JSON → SQLite## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Zdroje viditeľnosti pri spustení: +## 4) Cloud Sync Degradation -- protokoly konzoly z `src/sse/utils/logger.ts` -- agregáty využitia na požiadanie v SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- štvorstupňové podrobné zachytávanie užitočného zaťaženia v SQLite (`request_detail_logs`), keď je `settings.detailed_logs_enabled=true` -- textový denník stavu požiadavky v `log.txt` (voliteľné/kompatibilné) -- voliteľné hĺbkové protokoly požiadaviek/prekladov pod `logs/`, keď `ENABLE_REQUEST_LOGS=true` -- koncové body používania dashboardu (`/api/usage/*`) pre spotrebu používateľského rozhrania +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -Podrobné zachytenie dátovej časti požiadavky ukladá až štyri fázy užitočného zaťaženia JSON na jeden smerovaný hovor: +## 5) Data Integrity -- nespracovaná požiadavka prijatá od klienta -- preložená žiadosť skutočne odoslaná proti prúdu -- odpoveď poskytovateľa zrekonštruovaná ako JSON; streamované odpovede sú zhutnené do konečného súhrnu plus stream metadát -- konečná odpoveď klienta vrátená OmniRoute; streamované odpovede sú uložené v rovnakom kompaktnom súhrnnom formulári## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- Tajný kľúč JWT (`JWT_SECRET`) zabezpečuje overenie/podpísanie súboru cookie relácie dashboardu -- Zavedenie počiatočného hesla (`INITIAL_PASSWORD`) by malo byť explicitne nakonfigurované na poskytovanie pri prvom spustení -- Tajný kľúč API HMAC (`API_KEY_SECRET`) zabezpečuje vygenerovaný lokálny formát kľúča API -- Tajomstvá poskytovateľa (kľúče/tokeny API) sú uložené v lokálnej databáze a mali by byť chránené na úrovni súborového systému -- Koncové body cloudovej synchronizácie sa spoliehajú na sémantiku kľúča API + ID stroja## Environment and Runtime Matrix +## Observability and Operational Signals -Premenné prostredia aktívne používané kódom: +Runtime visibility sources: -- Aplikácia/autorizácia: `JWT_SECRET`, `INITIAL_PASSWORD` -- Úložisko: `DATA_DIR` -- Kompatibilné správanie uzla: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Voliteľné prepísanie základne úložiska (Linux/macOS, keď nie je nastavený `DATA_DIR`): `XDG_CONFIG_HOME` - – Hašovanie zabezpečenia: `API_KEY_SECRET`, `MACHINE_ID_SALT` - – Protokolovanie: „ENABLE_REQUEST_LOGS“. - – Synchronizácia/cloudové URL: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` - – Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` a varianty s malými písmenami -- Príznaky funkcie SOCKS5: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- Pomocníci platformy/behu (nie konfigurácia špecifická pre aplikáciu): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` a `localDb` zdieľajú rovnakú politiku základného adresára (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) s migráciou starších súborov. -2. `/api/v1/route.ts` deleguje rovnakého tvorcu jednotného katalógu, ktorý používa `/api/v1/models` (`src/app/api/v1/models/catalog.ts`), aby sa predišlo sémantickému posunu. -3. Požiadavka zapisovača zapíše úplné hlavičky/telo, keď je povolené; považovať adresár denníka za citlivý. -4. Správanie cloudu závisí od správnej adresy URL NEXT_PUBLIC_BASE_URL a od dostupnosti koncového bodu cloudu. -5. Adresár `open-sse/` je publikovaný ako balík pracovného priestoru `@omniroute/open-sse`**npm**. Zdrojový kód ho importuje cez `@omniroute/open-sse/...` (vyriešené Next.js `transpilePackages`). Cesty k súborom v tomto dokumente stále používajú názov adresára `open-sse/` kvôli konzistencii. -6. Grafy na ovládacom paneli používajú**Recharts**(založené na SVG) na prístupné interaktívne analytické vizualizácie (stĺpcové grafy používania modelov, tabuľky rozdelenia poskytovateľov s mierou úspešnosti). -7. E2E testy používajú**Playwright**(`tests/e2e/`), spúšťajú sa cez `npm run test:e2e`. Jednotkové testy používajú**Node.js test runner**(`tests/unit/`), spúšťajú sa cez `npm run test:unit`. Zdrojový kód pod `src/` je**TypeScript**(`.ts`/`.tsx`); pracovný priestor `open-sse/` zostáva JavaScript (`.js`). -8. Stránka s nastaveniami je usporiadaná do 5 záložiek: Zabezpečenie, Smerovanie (6 globálnych stratégií: fill-first, round-robin, p2c, náhodné, najmenej používané, nákladovo optimalizované), Resilience (upraviteľné limity sadzieb, istič, politiky), AI (rozpočet na myslenie, systémová výzva, prompt cache), Advanced (proxy).## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- Zostavte zo zdroja: `npm run build` -- Vytvorenie obrazu Docker: `docker build -t omniroute .` -- Spustite službu a overte: -- "ZÍSKAJTE /api/nastavenia". -- `ZÍSKAJTE /api/v1/models` -- Základná adresa URL cieľového CLI by mala byť `http://:20128/v1`, keď `PORT=20128` +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` +- `GET /api/v1/models` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/sk/docs/FEATURES.md b/docs/i18n/sk/docs/FEATURES.md index 2d222bd71a..02990dd96a 100644 --- a/docs/i18n/sk/docs/FEATURES.md +++ b/docs/i18n/sk/docs/FEATURES.md @@ -4,103 +4,168 @@ --- -Vizuálny sprievodca každou sekciou ovládacieho panela OmniRoute.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Spravujte pripojenia poskytovateľov AI: poskytovatelia OAuth (Claude Code, Codex, Gemini CLI), poskytovatelia kľúčov API (Groq, DeepSeek, OpenRouter) a bezplatní poskytovatelia (Qoder, Qwen, Kiro). Účty Kiro zahŕňajú sledovanie zostatku kreditu – zostávajúce kredity, celkový príspevok a dátum obnovenia sú viditeľné v Dashboard → Použitie.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Vytvorte kombá smerovania modelov so 6 stratégiami: prioritná, vážená, obojstranná, náhodná, najmenej používaná a nákladovo optimalizovaná. Každé kombo spája viacero modelov s automatickým vrátením a obsahuje rýchle šablóny a kontroly pripravenosti.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Komplexná analýza používania so spotrebou tokenov, odhadmi nákladov, teplotnými mapami aktivít, týždennými distribučnými grafmi a rozpismi podľa poskytovateľov.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Monitorovanie v reálnom čase: dostupnosť, pamäť, verzia, percentily latencie (p50/p95/p99), štatistiky vyrovnávacej pamäte a stavy ističov poskytovateľa.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Štyri režimy ladenia prekladov API:**Playground**(konvertor formátov), ​​**Chat Tester**(živé požiadavky),**Test Bench**(dávkové testy) a**Live Monitor**(stream v reálnom čase).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Otestujte akýkoľvek model priamo z palubnej dosky. Vyberte poskytovateľa, model a koncový bod, píšte výzvy pomocou editora Monaco, streamujte odpovede v reálnom čase, rušte uprostred streamu a zobrazujte metriky časovania.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Prispôsobiteľné farebné motívy pre celý prístrojový panel. Vyberte si zo 7 prednastavených farieb (koralová, modrá, červená, zelená, fialová, oranžová, azúrová) alebo si vytvorte vlastný motív výberom ľubovoľnej šesťhrannej farby. Podporuje svetlý, tmavý a systémový režim.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Komplexný panel nastavení s kartami: +Comprehensive settings panel with tabs: --**Všeobecné**— Systémové úložisko, správa zálohovania (export/import databázy) -**Vzhľad**— Výber motívu (tmavý/svetlý/systém), prednastavenia farebných motívov a vlastné farby, viditeľnosť zdravotného denníka, ovládacie prvky viditeľnosti položiek na bočnom paneli -**Bezpečnosť**— Ochrana koncového bodu API, blokovanie vlastného poskytovateľa, filtrovanie IP, informácie o relácii -**Routovanie**— Modelové aliasy, degradácia úloh na pozadí -–**Odolnosť**– Perzistencia rýchlostného limitu, ladenie ističa, automatické deaktivovanie zakázaných účtov, sledovanie uplynutia platnosti poskytovateľa -**Pokročilé**– Prepisy konfigurácie, záznam o audite konfigurácie, záložný režim degradácie![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Konfigurácia nástrojov na kódovanie AI jedným kliknutím: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor a Factory Droid. Obsahuje automatické použitie/resetovanie konfigurácie, profily pripojenia a mapovanie modelov.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Dashboard na vyhľadávanie a správu agentov CLI. Zobrazuje mriežku 14 vstavaných agentov (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) s: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Stav inštalácie**— Nainštalované / Nenájdené s detekciou verzie -**Odznaky protokolu**— stdio, HTTP atď. -**Vlastní agenti**— Zaregistrujte akýkoľvek nástroj CLI prostredníctvom formulára (názov, binárny súbor, príkaz verzie, spúšťacie argumenty) -**CLI Fingerprint Matching**– Prepínanie podľa jednotlivých poskytovateľov, aby sa zhodovali podpisy natívnych požiadaviek CLI, čím sa znižuje riziko zákazu pri zachovaní adresy IP proxy--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Vytvárajte obrázky, videá a hudbu z ovládacieho panela. Podporuje OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open a MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Protokolovanie požiadaviek v reálnom čase s filtrovaním podľa poskytovateľa, modelu, účtu a kľúča API. Zobrazuje stavové kódy, využitie tokenu, latenciu a podrobnosti o odozve.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Váš zjednotený koncový bod rozhrania API s rozdelením funkcií: Dokončenia rozhovoru, API odpovedí, vkladanie, generovanie obrázkov, zmena poradia, prepis zvuku, prevod textu na reč, moderovanie a registrované kľúče rozhrania API. Integrácia Cloudflare Quick Tunnel a podpora cloudového proxy pre vzdialený prístup.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -Vytváranie, rozsah a odvolávanie kľúčov API. Každý kľúč môže byť obmedzený na konkrétne modely/poskytovateľov s plným prístupom alebo oprávneniami len na čítanie. Vizuálna správa kľúčov so sledovaním používania.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Sledovanie administratívnej akcie s filtrovaním podľa typu akcie, aktéra, cieľa, IP adresy a časovej pečiatky. Úplná história bezpečnostných udalostí.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Natívna desktopová aplikácia Electron pre Windows, MacOS a Linux. Spustite OmniRoute ako samostatnú aplikáciu s integráciou na systémovej lište, offline podporou, automatickou aktualizáciou a inštaláciou jedným kliknutím. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Kľúčové vlastnosti: +Key features: -- Dotazovanie pripravenosti servera (žiadna prázdna obrazovka pri studenom štarte) -- Systémová lišta so správou portov -- Zásady zabezpečenia obsahu -- Jednostupňový zámok +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock - Auto-update on restart -- Platformovo podmienené používateľské rozhranie (semafory pre macOS, predvolený nadpis systému Windows/Linux) -- Balenie zostavy zosilnených elektrónov – symbolicky prepojené `node_modules` v samostatnom balíku sa pred balením detegujú a odmietnu, čím sa zabráni závislosti spustenia od zostavovacieho stroja (v2.5.5+) +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Úplnú dokumentáciu nájdete v [`electron/README.md`](../electron/README.md). +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/sk/docs/TROUBLESHOOTING.md b/docs/i18n/sk/docs/TROUBLESHOOTING.md index 7ce76bffeb..308b8d379a 100644 --- a/docs/i18n/sk/docs/TROUBLESHOOTING.md +++ b/docs/i18n/sk/docs/TROUBLESHOOTING.md @@ -4,69 +4,142 @@ --- -Bežné problémy a riešenia pre OmniRoute.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Problém | Riešenie | -| ----------------------------------------------- | --------------------------------------------------------------------------- | --- | -| Prvé prihlásenie nefunguje | Nastaviť `INITIAL_PASSWORD` v `.env` (žiadne napevno zakódované predvolené) | -| Prístrojová doska sa otvára na nesprávnom porte | Nastaviť `PORT=20128` a `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Žiadne záznamy žiadostí pod `logs/` | Nastavte `ENABLE_REQUEST_LOGS=true` | -| EACCES: povolenie zamietnuté | Nastavte `DATA_DIR=/path/to/writable/dir` na prepísanie `~/.omniroute` | -| Stratégia smerovania sa neukladá | Aktualizácia na v1.4.11+ (Oprava schémy Zod pre pretrvávanie nastavení) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Príčina:**Kvóta poskytovateľa je vyčerpaná. +**Cause:** Provider quota exhausted. -**Oprava:** +**Fix:** -1. Skontrolujte sledovanie kvót palubnej dosky -2. Použite kombináciu so záložnými vrstvami -3. Prejdite na lacnejšiu/bezplatnú úroveň### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Príčina:**Kvóta odberov je vyčerpaná. +### Rate Limiting -**Oprava:** +**Cause:** Subscription quota exhausted. -– Pridajte záložný kód: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +**Fix:** -- Použite GLM/MiniMax ako lacnú zálohu### OAuth Token Expired +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -OmniRoute automaticky obnovuje tokeny. Ak problémy pretrvávajú: +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: 1. Dashboard → Provider → Reconnect -2. Odstráňte a znova pridajte pripojenie poskytovateľa--- +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Overte, či „BASE_URL“ odkazuje na vašu spustenú inštanciu (napr. „http://localhost:20128“) -2. Overte, že `CLOUD_URL` odkazuje na váš koncový bod cloudu (napr. `https://omniroute.dev`) -3. Ponechajte hodnoty „NEXT*PUBLIC*\*“ zarovnané s hodnotami na strane servera### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Príznak:**`Neočakávaný token 'd'...` na koncovom bode cloudu pre hovory bez streamovania. +### Cloud `stream=false` Returns 500 -**Príčina:**Upstream vracia užitočné zaťaženie SSE, zatiaľ čo klient očakáva JSON. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Náhradné riešenie:**Pre priame hovory v cloude použite `stream=true`. Lokálne prostredie runtime zahŕňa záložnú verziu SSE→JSON.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Vytvorte nový kľúč z miestneho informačného panela (`/api/keys`) -2. Spustite synchronizáciu s cloudom: Povoliť cloud → Synchronizovať teraz -3. Staré/nesynchronizované kľúče môžu v cloude stále vrátiť „401“.--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Skontrolujte runtime polia: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. Pre prenosný režim: použite image target `runner-cli` (pribalené CLI) -3. Pre režim pripojenia hostiteľa: nastavte `CLI_EXTRA_PATHS` a pripojte adresár bin hostiteľa ako iba na čítanie -4. Ak `installed=true` a `runnable=false`: binárny súbor bol nájdený, ale zlyhala kontrola stavu### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -80,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Skontrolujte štatistiky používania v Dashboard → Usage -2. Prepnite primárny model na GLM/MiniMax -3. Na nekritické úlohy používajte bezplatnú vrstvu (Gemini CLI, Qoder). -4. Nastavte rozpočty nákladov na kľúč API: Dashboard → API Keys → Budget--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Vo svojom súbore .env nastavte hodnotu `ENABLE_REQUEST_LOGS=true`. Protokoly sa zobrazujú v adresári `logs/`.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -101,105 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Hlavný stav: `${DATA_DIR}/storage.sqlite` (poskytovatelia, kombá, aliasy, kľúče, nastavenia) -- Použitie: tabuľky SQLite v `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + voliteľné `${DATA_DIR}/log.txt` a `${DATA_DIR}/call_logs/` -- Denníky žiadostí: `/logs/...` (keď `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Keď je istič poskytovateľa OTVORENÝ, požiadavky sú zablokované, kým nevyprší cooldown. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Oprava:** +**Fix:** -1. Prejdite na**Hlavný panel → Nastavenia → Odolnosť** -2. Skontrolujte kartu ističa príslušného poskytovateľa -3. Kliknite na**Reset All**, aby ste vymazali všetky ističe, alebo počkajte, kým uplynie cooldown -4. Pred resetovaním skontrolujte, či je poskytovateľ skutočne dostupný### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Ak poskytovateľ opakovane prejde do stavu OTVORENÉ: +### Provider keeps tripping the circuit breaker -1. Vzor zlyhania nájdete v**Dashboard → Health → Provider Health** -2. Prejdite na**Nastavenia → Odolnosť → Profily poskytovateľa**a zvýšte prah zlyhania -3. Skontrolujte, či poskytovateľ zmenil limity API alebo či nevyžaduje opätovné overenie -4. Skontrolujte telemetriu latencie – vysoká latencia môže spôsobiť zlyhania súvisiace s časovým limitom--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Uistite sa, že používate správnu predponu: `deepgram/nova-3` alebo `assemblyai/best` - – Overte, či je poskytovateľ pripojený v**Dashboard → Providers**### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Skontrolujte podporované zvukové formáty: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- Overte, či je veľkosť súboru v rámci limitov poskytovateľa (zvyčajne < 25 MB) -- Skontrolujte platnosť kľúča API poskytovateľa na karte poskytovateľa--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Na ladenie problémov s prekladom formátu použite**Dashboard → Translator**: +Use **Dashboard → Translator** to debug format translation issues: -| Režim | Kedy použiť | -| --------------------- | --------------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Ihrisko** | Porovnajte vstupné/výstupné formáty vedľa seba — prilepte neúspešnú požiadavku, aby ste videli, ako sa prekladá | -| **Tester chatu** | Posielajte živé správy a skontrolujte celý obsah žiadosti/odpovede vrátane hlavičiek | -| **Testovacia lavica** | Spustite dávkové testy kombinácií formátov, aby ste zistili, ktoré preklady sú poškodené | -| **Živý monitor** | Sledujte tok žiadostí v reálnom čase, aby ste zachytili občasné problémy s prekladom | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Značky myslenia sa nezobrazujú**— Skontrolujte, či cieľový poskytovateľ podporuje myslenie a nastavenie rozpočtu na myslenie -**Volania nástrojov klesajú**– Niektoré preklady formátov môžu odstrániť nepodporované polia; overiť v režime Playground -**Chýba systémová výzva**– Claude a Gemini riešia výzvy systému odlišne; skontrolujte výstup prekladu -–**SDK vracia nespracovaný reťazec namiesto objektu**– Opravené vo verzii 1.1.0: nástroj na dezinfekciu odpovede teraz odstraňuje neštandardné polia (`x_groq`, `usage_breakdown` atď.), ktoré spôsobujú zlyhania overenia OpenAI SDK Pydantic -**GLM/ERNIE odmieta `systémovú` rolu**— Opravené vo verzii 1.1.0: normalizátor rolí automaticky spája systémové správy do užívateľských správ pre nekompatibilné modely +### Common format issues -- Rola**`vývojára` nie je rozpoznaná**— Opravené vo verzii 1.1.0: automaticky prevedené na `systém` pre poskytovateľov, ktorí nie sú OpenAI -**`json_schema` nefunguje s Gemini**– Opravené vo verzii 1.1.0: `response_format` je teraz skonvertovaný na `responseMimeType` + `responseSchema` Gemini--- +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- Automatický limit sadzby sa vzťahuje len na poskytovateľov kľúčov API (nie OAuth/predplatné) -- Skontrolujte, či je v**Nastaveniach → Odolnosť → Profily poskytovateľov**povolený automatický limit rýchlosti -- Skontrolujte, či poskytovateľ vracia stavové kódy `429` alebo hlavičky `Retry-After`### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -Profily poskytovateľov podporujú tieto nastavenia: +### Tuning exponential backoff --**Základné oneskorenie**— Počiatočná doba čakania po prvom zlyhaní (predvolené: 1 s) -–**Maximálne oneskorenie**– Obmedzenie maximálnej doby čakania (predvolené: 30 s) -**Násobiteľ**– o koľko sa má predĺžiť oneskorenie pri následnom zlyhaní (predvolené: 2x)### Anti-thundering herd +Provider profiles support these settings: -Keď mnoho súbežných požiadaviek zasiahne poskytovateľa s obmedzenou rýchlosťou, OmniRoute použije mutex + automatické obmedzenie rýchlosti na serializáciu požiadaviek a zabránenie kaskádovým zlyhaniam. Toto je automatické pre poskytovateľov kľúčov API.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Niektorí používatelia OmniRoute umiestňujú bránu pred zásobníky RAG alebo agentov. V týchto nastaveniach je bežné vidieť zvláštny vzor: OmniRoute vyzerá zdravo (poskytovatelia sú v poriadku, smerovacie profily sú v poriadku, žiadne upozornenia na obmedzenie rýchlosti), ale konečná odpoveď je stále nesprávna. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -V praxi tieto incidenty zvyčajne pochádzajú z dolného potrubia RAG, nie zo samotnej brány. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Ak chcete mať zdieľaný slovník na popis týchto zlyhaní, môžete použiť WFGY ProblemMap, externý textový zdroj licencie MIT, ktorý definuje šestnásť opakujúcich sa vzorov porúch RAG / LLM. Na vysokej úrovni pokrýva: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- posun pri vyhľadávaní a narušené hranice kontextu -- prázdne alebo zastarané indexy a vektorové sklady -- vkladanie verzus sémantický nesúlad -- rýchle zostavenie a problémy s kontextovým oknom -- logický kolaps a príliš sebavedomé odpovede -- zlyhanie koordinácie dlhých reťazcov a agentov -- multiagentová pamäť a posun rolí -- problémy s nasadením a objednávaním bootstrapu +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -Myšlienka je jednoduchá: +The idea is simple: -1. Keď preskúmate zlú odpoveď, zaznamenajte: - - užívateľská úloha a požiadavka - - kombinácia trasy alebo poskytovateľa v OmniRoute - - akýkoľvek kontext RAG použitý nadol (získané dokumenty, volania nástrojov atď.) -2. Priraďte incident k jednému alebo dvom číslam WFGY ProblemMap (`č.1` … `č.16`). -3. Uložte si číslo na svoj vlastný informačný panel, runbook alebo sledovač incidentov vedľa protokolov OmniRoute. -4. Pomocou príslušnej stránky WFGY sa rozhodnite, či potrebujete zmeniť stratégiu zásobníka RAG, retrievera alebo smerovania. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -Celý text a konkrétne recepty nájdete tu (licencia MIT, len text): +Full text and concrete recipes live here (MIT license, text only): -[README WFGY ProblemMap](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) +[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -Túto sekciu môžete ignorovať, ak za OmniRoute nespúšťate RAG alebo agentov.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? -–**Problémy s GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Architektúra**: Interné podrobnosti nájdete v [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) -**Referencia API**: Všetky koncové body nájdete v [`docs/API_REFERENCE.md`](API_REFERENCE.md) -**Hlavný panel zdravia**: Skontrolujte stav systému v reálnom čase v**Hlavnom paneli → Zdravie** -**Prekladač**: Na ladenie problémov s formátom použite**Dashboard → Translator** +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/sk/llm.txt b/docs/i18n/sk/llm.txt new file mode 100644 index 0000000000..d67e414ece --- /dev/null +++ b/docs/i18n/sk/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Slovenčina) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Prehľad + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Bezpečnosť +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/sv/README.md b/docs/i18n/sv/README.md index db437b3779..b41860c18d 100644 --- a/docs/i18n/sv/README.md +++ b/docs/i18n/sv/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Din universella API-proxy — en slutpunkt, 60+ leverantörer, noll driftstopp. Nu med**MCP Server (25 verktyg)**,**A2A Protocol**,**Memory/Skills Systems**och**Electron Desktop App**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Slutförda chattar • Inbäddningar • Bildgenerering • Video • Musik • Ljud • Omrankning •**Webbsökning**• MCP-server • A2A-protokoll • 100 % TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Din universella API-proxy — en slutpunkt, 60+ leverantörer, noll driftstopp. [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Webbplats](https://omniroute.online) • [🚀 Snabbstart](#-quick-start) • [💡 Funktioner](#-key-features) • [📖 Dokument](#-dokumentation) • [💰 Prissättning](#-pricing-at-a-llance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Tillgänglig på:**🇺🇸 [engelska](README.md) | 🇧🇷 [Português (Brasilien)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [tyska](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [norska](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [filippinska](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,28 +60,30 @@ _Din universella API-proxy — en slutpunkt, 60+ leverantörer, noll driftstopp. ## 📸 Dashboard Preview - -Klicka för att se skärmdumpar från instrumentpanelen +
+Click to see dashboard screenshots -| Sida | Skärmdump | -| --------------------- | -------------------------------------------------- | ---------- | -| **Leverantörer** | ![Providers](docs/screenshots/01-providers.png) | -| **Kombos** | ![Combos](docs/screenshots/02-combos.png) | -| **Analytik** | ![Analytics](docs/screenshots/03-analytics.png) | -| **Hälsa** | ![Hälsa](docs/screenshots/04-health.png) | -| **Översättare** | ![Översättare](docs/screenshots/05-translator.png) | -| **Inställningar** | ![Settings](docs/screenshots/06-settings.png) | -| **CLI-verktyg** | ![CLI-verktyg](docs/screenshots/07-cli-tools.png) | -| **Användningsloggar** | ![Användning](docs/screenshots/08-usage.png) | -| **Slutpunkter** | ![Endpoints](docs/screenshots/09-endpoint.png) |
| +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | + + --- ### 🤖 Free AI Provider for your favorite coding agents -_Anslut alla AI-drivna IDE- eller CLI-verktyg via OmniRoute — gratis API-gateway för obegränsad kodning._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ - + @@ -125,481 +134,555 @@ _Anslut alla AI-drivna IDE- eller CLI-verktyg via OmniRoute — gratis API-gatew Codex CLI
Codex CLI
- ⭐ 60,8K + ⭐ 60.8K
@@ -88,28 +97,28 @@ _Anslut alla AI-drivna IDE- eller CLI-verktyg via OmniRoute — gratis API-gatew NanoBot
NanoBot

- ⭐ 20,9K + ⭐ 20.9K
PicoClaw
PicoClaw

- ⭐ 14,6K + ⭐ 14.6K
ZeroClaw
ZeroClaw

- ⭐ 9,9K + ⭐ 9.9K
IronClaw
IronClaw

- ⭐ 2,1K + ⭐ 2.1K
Claude Code
Claude Code

- ⭐ 67,3K + ⭐ 67.3K
Gemini CLI
Gemini CLI

- ⭐ 94,7K + ⭐ 94.7K
- Kilokod
- Kilokod + Kilo Code
+ Kilo Code

- ⭐ 15,5K + ⭐ 15.5K
-📡 Alla agenter ansluter via http://localhost:20128/v1 eller http://cloud.omniroute.online/v1 — en konfiguration, obegränsade modeller och kvot--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Sluta slösa pengar och nå gränser:** +**Stop wasting money and hitting limits:** -- Prenumerationskvoten löper ut oanvänd varje månad -- Prisgränser stoppar dig mellankodning -- Dyra API:er ($20-50/månad per leverantör) -- Manuellt byte mellan leverantörer +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute löser detta:** +**OmniRoute solves this:** -- ✅**Maximera prenumerationer**- Spåra kvot, använd varje bit innan återställning -- ✅**Automatisk reserv**- Prenumeration → API-nyckel → Billigt → Gratis, noll driftstopp -- ✅**Multi-konto**- Round-robin mellan konton per leverantör -- ✅**Universal**- Fungerar med Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, vilket CLI-verktyg som helst--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Gå med i vår community!**[WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Få hjälp, dela tips och håll dig uppdaterad. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Webbplats**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Problem**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Bidrar**: Se [CONTRIBUTING.md](CONTRIBUTING.md), öppna en PR eller välj ett "bra första nummer" -**Original Project**: [9router av decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -När du öppnar ett problem, kör kommandot systeminfo och bifoga den genererade filen:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Detta genererar en `system-info.txt` med din Node.js-version, OmniRoute-version, OS-detaljer, installerade CLI-verktyg (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2-status och systempaket - allt vi behöver för att snabbt reproducera ditt problem. Bifoga filen direkt till ditt GitHub-problem.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Varje utvecklare som använder AI-verktyg möter dessa problem dagligen.**OmniRoute byggdes för att lösa dem alla — från kostnadsöverskridanden till regionala block, från trasiga OAuth-flöden till protokolloperationer och observerbarhet i företag. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. - -💸 1. "Jag betalar för en dyr prenumeration men blir ändå avbruten av limits" +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Utvecklare betalar $20–200/månad för Claude Pro, Codex Pro eller GitHub Copilot. Även om du betalar har kvoten ett tak - 5 timmars användning, veckogränser eller gränser per minut. Mid-coding session, leverantören slutar svara och utvecklaren tappar flöde och produktivitet. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Hur OmniRoute löser det:** +**How OmniRoute solves it:** --**Smart 4-lagers fallback**— Om prenumerationskvoten tar slut, omdirigeras automatiskt till API-nyckel → Billigt → Gratis med noll manuellt ingrepp --**Spårning av leverantörsgränser**— Cachade ögonblicksbilder av kvoter uppdateras på ett schema på serversidan (standard `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) med manuell uppdatering tillgänglig i användargränssnittet --**Stöd för flera konton**- Flera konton per leverantör med automatisk round-robin - när ett tar slut, byter du till nästa --**Anpassade kombinationer**— Anpassningsbara reservkedjor med 9 balanseringsstrategier (prioritet, viktad, fyll först, round-robin, P2C, slumpmässig, minst använda, kostnadsoptimerad, strikt slumpmässig) --**Codex Business Quotas**— Övervakning av företags-/teamarbetsutrymmeskvoter direkt i instrumentpanelen
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard - -🔌 2. "Jag måste använda flera leverantörer men alla har olika API" + -OpenAI använder ett format, Claude (Anthropic) använder ett annat, Gemini ännu ett annat. Om en utvecklare vill testa modeller från olika leverantörer eller fallback mellan dem måste de konfigurera om SDK:er, ändra slutpunkter, hantera inkompatibla format. Anpassade leverantörer (FriendLI, NIM) har icke-standardiserade modellslutpunkter. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Hur OmniRoute löser det:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Unified Endpoint**— En enda "http://localhost:20128/v1" fungerar som proxy för alla 60+ leverantörer --**Formatöversättning**— Automatisk och transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API --**Response Sanitization**— Tar bort icke-standardiserade fält (`x_groq`, `usage_breakdown`, `service_tier`) som bryter OpenAI SDK v1.83+ --**Rollnormalisering**— Konverterar `utvecklare` → `system` för icke-OpenAI-leverantörer; `system` → `användare` för GLM/ERNIE --**Think Tag Extraction**— Extraherar ``-block från modeller som DeepSeek R1 till standardiserat `reasoning_content` --**Structured Output for Gemini**— `json_schema` → `responseMimeType`/`responseSchema` automatisk konvertering --**`stream` har som standard "false"**- Justerar med OpenAI-specifikationen, undviker oväntade SSE i Python/Rust/Go SDK:er
+**How OmniRoute solves it:** - -🌐 3. "Min AI-leverantör blockerar min region/land" +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Leverantörer som OpenAI/Codex blockerar åtkomst från vissa geografiska regioner. Användare får fel som "unsupported_country_region_territory" under OAuth- och API-anslutningar. Detta är särskilt frustrerande för utvecklare från utvecklingsländer. + -**Hur OmniRoute löser det:** +
+🌐 3. "My AI provider blocks my region/country" --**3-Level Proxy Config**— Konfigurerbar proxy på 3 nivåer: global (all trafik), per leverantör (endast en leverantör) och per anslutning/nyckel --**Färgkodade proxymärken**— Visuella indikatorer: 🟢 global proxy, 🟡 leverantörsproxy, 🔵 anslutningsproxy, visar alltid IP:n --**OAuth Token Exchange Through Proxy**— OAuth-flödet går också genom proxyn och löser "unsupported_country_region_territory" --**Anslutningstester via proxy**— Anslutningstester använder den konfigurerade proxyn (ingen mer direkt förbikoppling) --**SOCKS5-stöd**— Fullständigt SOCKS5-proxystöd för utgående routing --**TLS Fingerprint Spoofing**— Webbläsarliknande TLS-fingeravtryck via "wreq-js" för att kringgå botdetektering --**🔏 CLI Fingerprint Matching**— Ordnar om rubriker och kroppsfält för att matcha inbyggda CLI-binära signaturer, vilket drastiskt minskar risken för kontoflaggning. Proxy-IP:n bevaras – du får både smyg**och**IP-maskering samtidigt
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. - -🆓 4. "Jag vill använda AI för kodning men jag har inga pengar" +**How OmniRoute solves it:** -Alla kan inte betala $20–200/månad för AI-prenumerationer. Studenter, utvecklare från tillväxtländer, hobbyister och frilansare behöver tillgång till kvalitetsmodeller utan kostnad. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Hur OmniRoute löser det:** + --**Gratis leverantörer inbyggda**— Inbyggt stöd för 100 % gratis leverantörer: Qoder (5 obegränsade modeller via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited-modeller:-r-modeller:-r qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID gratis), Gemini CLI (180K tokens/månad gratis) --**Ollama Cloud**— Ollama-modeller med molnvärd på `api.ollama.com` med gratis nivå "Lätt användning"; använd prefixet `ollamacloud/` --**Free-Only Combos**— Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/månad utan stilleståndstid --**NVIDIA NIM fri tillgång**— ~40 RPM dev-för evigt fri tillgång till 70+ modeller på build.nvidia.com (övergång från krediter till rena hastighetsgränser) --**Kostnadsoptimerad strategi**— Routingstrategi som automatiskt väljer den billigaste tillgängliga leverantören +
+🆓 4. "I want to use AI for coding but I have no money" - -🔒 5. "Jag måste skydda min AI-gateway från obehörig åtkomst" +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -När du exponerar en AI-gateway för nätverket (LAN, VPS, Docker) kan vem som helst med adressen konsumera utvecklarens tokens/kvot. Utan skydd är API:er sårbara för missbruk, snabb injektion och missbruk. +**How OmniRoute solves it:** -**Hur OmniRoute löser det:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**API Key Management**— Generering, rotation och omfattning per leverantör med en dedikerad `/dashboard/api-manager`-sida --**Behörigheter på modellnivå**— Begränsa API-nycklar till specifika modeller ('openai/*', jokerteckenmönster), med växlingen Tillåt alla/Begränsa --**API Endpoint Protection**— Kräv en nyckel för `/v1/modeller` och blockera specifika leverantörer från listan --**Auth Guard + CSRF Protection**— Alla instrumentpanelsrutter skyddade med 'withAuth' middleware + CSRF-tokens --**Rate Limiter**— Per-IP-hastighetsbegränsning med konfigurerbara fönster --**IP-filtrering**— Tillåtelselista/blockeringslista för åtkomstkontroll --**Prompt Injection Guard**— Sanering mot skadliga promptmönster --**AES-256-GCM-kryptering**— Autentiseringsuppgifter krypterade i vila
+ - -🛑 6. "Min leverantör gick ner och jag tappade mitt kodningsflöde" +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -AI-leverantörer kan bli instabila, returnera 5xx-fel eller nå tillfälliga hastighetsgränser. Om en utvecklare är beroende av en enskild leverantör avbryts de. Utan strömbrytare kan upprepade försök krascha programmet. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Hur OmniRoute löser det:** +**How OmniRoute solves it:** --**Circuit Breaker per modell**— Autoöppning/stängning med konfigurerbara trösklar och nedkylning (stängd/öppen/halvöppen), omfattning per modell för att undvika kaskadblock --**Exponentiell backoff**— Progressiva fördröjningar igen --**Anti-Thundering Herd**— Mutex + semaforskydd mot samtidiga stormar igen --**Combo reservkedjor**— Om den primära leverantören misslyckas, faller den automatiskt genom kedjan utan ingrepp --**Combo Circuit Breaker**- Inaktiverar automatiskt felande leverantörer inom en kombinationskedja --**Health Dashboard**— Drifttidsövervakning, strömbrytartillstånd, låsningar, cachestatistik, p50/p95/p99 latens
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest - -🔧 7. "Att konfigurera varje AI-verktyg är tråkigt och repetitivt" + -Utvecklare använder Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Varje verktyg behöver en annan konfiguration (API-slutpunkt, nyckel, modell). Att konfigurera om när man byter leverantör eller modell är ett slöseri med tid. +
+🛑 6. "My provider went down and I lost my coding flow" -**Hur OmniRoute löser det:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**CLI Tools Dashboard**— Dedikerad sida med ett-klicksinställningar för Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline --**GitHub Copilot Config Generator**— Genererar `chatLanguageModels.json` för VS-kod med bulkmodellval --**Onboarding Wizard**— Guidad 4-stegs installation för förstagångsanvändare --**En slutpunkt, alla modeller**— Konfigurera "http://localhost:20128/v1" en gång, åtkomst till 60+ leverantörer
+**How OmniRoute solves it:** - -🔑 8. "Hantera OAuth-tokens från flera leverantörer är ett helvete" +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot — alla använder OAuth 2.0 med utgående tokens. Utvecklare måste autentisera om hela tiden, hantera "klienthemlighet saknas", "redirect_uri_mismatch" och fel på fjärrservrar. OAuth på LAN/VPS är särskilt problematiskt. + -**Hur OmniRoute löser det:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Automatisk uppdatering av token**— OAuth-tokens uppdateras i bakgrunden innan de löper ut --**OAuth 2.0 (PKCE) Inbyggd**— Automatiskt flöde för Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder --**Multi-Account OAuth**— Flera konton per leverantör via JWT/ID-tokenextraktion --**OAuth LAN/Remote Fix**— Privat IP-detektering för `redirect_uri` + manuellt URL-läge för fjärrservrar --**OAuth Behind Nginx**— Använder `window.location.origin` för omvänd proxykompatibilitet --**Remote OAuth Guide**— Steg-för-steg-guide för Google Cloud-uppgifter på VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. - -📊 9. "Jag vet inte hur mycket jag spenderar eller var" +**How OmniRoute solves it:** -Utvecklare använder flera betalleverantörer men har ingen enhetlig syn på utgifter. Varje leverantör har sin egen faktureringspanel, men det finns ingen konsoliderad vy. Oväntade kostnader kan hopa sig. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Hur OmniRoute löser det:** + --**Kostnadsanalysinstrumentpanel**— Kostnadsspårning per token och budgethantering per leverantör --**Budgetgränser per nivå**— Utgiftstak per nivå som utlöser automatisk reserv --**Priskonfiguration per modell**— Konfigurerbara priser per modell --**Användningsstatistik per API-nyckel**— Antal förfrågningar och senast använda tidsstämpel per nyckel --**Analytics Dashboard**— Statistikkort, modellanvändningsdiagram, leverantörstabell med framgångsfrekvens och latens +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" - -🐛 10. "Jag kan inte diagnostisera fel och problem i AI-samtal" +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -När ett samtal misslyckas vet inte utvecklaren om det var en hastighetsgräns, utgången token, fel format eller leverantörsfel. Fragmenterade loggar över olika terminaler. Utan observerbarhet är felsökning att trial-and-error. +**How OmniRoute solves it:** -**Hur OmniRoute löser det:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Unified Logs Dashboard**— 4 flikar: Request Logs, Proxy Logs, Audit Logs, Console --**Console Log Viewer**— Viewer i realtid i terminalstil med färgkodade nivåer, automatisk rullning, sökning, filtrering --**SQLite Proxy-loggar**— Beständiga loggar som överlever serverstarter --**Translator Playground**— 4 felsökningslägen: Playground (formatöversättning), Chat Tester (tur och retur), Testbänk (batch), Live Monitor (realtid) --**Request Telemetri**— p50/p95/p99 latens + X-Request-Id-spårning --**Filbaserad loggning med rotation**— Apploggar roterar efter storlek, lagringsdagar och arkivantal; anropsloggartefakter roterar efter lagringsdagar och filantal --**System Info Report**— `npm run system-info` genererar `system-info.txt` med din fullständiga miljö (nodversion, OmniRoute-version, OS, CLI-verktyg, Docker/PM2-status). Bifoga den när du rapporterar problem för omedelbar triage.
+ - -🏗️ 11. "Det är komplicerat att distribuera och underhålla gatewayen" +
+📊 9. "I don't know how much I'm spending or where" -Att installera, konfigurera och underhålla en AI-proxy i olika miljöer (lokalt, VPS, Docker, moln) är arbetskrävande. Problem som hårdkodade sökvägar, "EACCES" på kataloger, portkonflikter och plattformsoberoende konstruktioner ger friktion. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Hur OmniRoute löser det:** +**How OmniRoute solves it:** --**npm global installation**— `npm install -g omniroute && omniroute` – klar --**Docker Multi-Platform**— AMD64 + ARM64 inbyggt (Apple Silicon, AWS Graviton, Raspberry Pi) --**Docker Compose Profiles**— 'base' (inga CLI-verktyg) och 'cli' (med Claude Code, Codex, OpenClaw) --**Electron Desktop App**— Inbyggd app för Windows/macOS/Linux med systemfältet, autostart, offlineläge --**Split-Port Mode**— API och Dashboard på separata portar för avancerade scenarier (omvänd proxy, containernätverk) --**Cloud Sync**— Konfigurera synkronisering mellan enheter via Cloudflare Workers --**DB-säkerhetskopior**— Automatisk säkerhetskopiering, återställning, export och import av alla inställningar, med `DISABLE_SQLITE_AUTO_BACKUP` för externt hanterade säkerhetskopior
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency - -🌍 12. "Gränssnittet är endast engelska och mitt team talar inte engelska" + -Lag i icke-engelsktalande länder, särskilt i Latinamerika, Asien och Europa, kämpar med enbart engelska gränssnitt. Språkbarriärer minskar användningen och ökar konfigurationsfelen. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Hur OmniRoute löser det:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Dashboard i18n — 30 språk**— Alla 500+ nycklar översatta, inklusive arabiska, bulgariska, danska, tyska, spanska, finska, franska, hebreiska, hindi, ungerska, indonesiska, italienska, japanska, koreanska, malaysiska, holländska, norska, polska, portugisiska (PT/BR), rumänska, ryska, thailändska, ukrainska, ukrainska, kinesiska, engelska, ukrainska, vietnamesiska, ukrainska, svenska, ukrainska --**RTL-stöd**— Höger-till-vänster-stöd för arabiska och hebreiska --**Multi-Language READMEs**— 30 fullständiga dokumentationsöversättningar --**Språkväljare**— Globikon i rubriken för växling i realtid
+**How OmniRoute solves it:** - -🔄 13. "Jag behöver mer än chatt — jag behöver inbäddningar, bilder, ljud" +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -AI är inte bara att slutföra chatt. Utvecklare måste generera bilder, transkribera ljud, skapa inbäddningar för RAG, ranka om dokument och moderera innehåll. Varje API har olika slutpunkt och format. + -**Hur OmniRoute löser det:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Inbäddningar**— `/v1/inbäddningar` med 6 leverantörer och 9+ modeller --**Bildgenerering**— `/v1/images/generations` med 10 leverantörer och 20+ modeller (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Text-to-Video**— `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) och SD WebUI --**Text-to-Music**— `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) --**Audio Transcription**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Text-to-Speech**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + befintliga leverantörer --**Moderations**— `/v1/moderations` — Innehållssäkerhetskontroller --**Omrankning**— `/v1/rerank` — Omrankning av dokumentrelevans --**Responses API**— Fullständigt `/v1/responses`-stöd för Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. - -🧪 14. "Jag har inget sätt att testa och jämföra kvalitet mellan olika modeller" +**How OmniRoute solves it:** -Utvecklare vill veta vilken modell som är bäst för deras användningsfall - kod, översättning, resonemang - men det går långsamt att jämföra manuellt. Det finns inga integrerade utvärderingsverktyg. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**Hur OmniRoute löser det:** + --**LLM Evaluations**— Golden set-testning med 10 förinstallerade fall som täcker hälsningar, matematik, geografi, kodgenerering, JSON-efterlevnad, översättning, markdown, säkerhetsvägran --**4 matchningsstrategier**— "exakt", "innehåller", "regex", "anpassad" (JS-funktion) --**Translator Playground Test Bench**— Batchtestning med flera ingångar och förväntade utgångar, jämförelse mellan olika leverantörer --**Chatttestare**— Fullständig tur och retur med visuell responsåtergivning --**Live Monitor**— Realtidsström av alla förfrågningar som flödar genom proxyn +
+🌍 12. "The interface is English-only and my team doesn't speak English" - -📈 15. "Jag behöver skala utan att förlora prestanda" +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -När förfrågningsvolymen ökar, utan att cachelagra genererar samma frågor dubbla kostnader. Utan idempotens, dubbletter begär avfallshantering. Prisgränser per leverantör måste respekteras. +**How OmniRoute solves it:** -**Hur OmniRoute löser det:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Semantisk cache**— Tvåskiktscache (signatur + semantisk) minskar kostnaden och fördröjningen --**Request Idempotency**— 5s dedupliceringsfönster för identiska förfrågningar --**Rate Limit Detection**— RPM per leverantör, min gap och max samtidig spårning --**Redigerbara hastighetsgränser**— Konfigurerbara standardinställningar i Inställningar → Motståndskraft med uthållighet --**API Key Validation Cache**— 3-lagers cache för produktionsprestanda --**Hälsoinstrumentpanel med telemetri**— p50/p95/p99 latens, cachestatistik, drifttid
+ - -🤖 16. "Jag vill kontrollera modellbeteende globalt" +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Utvecklare som vill ha alla svar på ett specifikt språk, med en specifik ton, eller som vill begränsa resonemangstokens. Att konfigurera detta i varje verktyg/förfrågan är opraktiskt. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Hur OmniRoute löser det:** +**How OmniRoute solves it:** --**System Prompt Injection**— Global prompt tillämpas på alla förfrågningar --**Thinking Budget Validation**— Reasoning token allocation control per request (passthrough, auto, custom, adaptive) --**9 routingstrategier**— Globala strategier som avgör hur förfrågningar distribueras --**Wildcard Router**— `provider/*`-mönster dirigerar dynamiskt till vilken leverantör som helst --**Kombo Aktivera/Inaktivera Växla**— Växla kombinationer direkt från instrumentpanelen --**Visa leverantör**— Aktivera/inaktivera alla anslutningar för en leverantör med ett klick --**Blockerade leverantörer**— Exkludera specifika leverantörer från `/v1/models`-listan
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex - -🧰 17. "Jag behöver MCP-verktyg som förstklassiga produktegenskaper" + -Många AI-gateways exponerar MCP endast som en dold implementeringsdetalj. Team behöver ett synligt, hanterbart driftlager. +
+🧪 14. "I have no way to test and compare quality across models" -**Hur OmniRoute löser det:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP visas på navigeringspanelen och fliken för slutpunktsprotokoll -- Dedikerad MCP-hanteringssida med process, verktyg, omfattningar och revision -- Inbyggd snabbstart för `omniroute --mcp` och klientintroduktion
+**How OmniRoute solves it:** - -🧠 18. "Jag behöver A2A-orkestrering med sökvägar för synkronisering och streaming" +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Agentarbetsflöden kräver både direkta svar och långvarig streamad exekvering med livscykelkontroll. + -**Hur OmniRoute löser det:** +
+📈 15. "I need to scale without losing performance" -- A2A JSON-RPC-slutpunkt ('POST /a2a') med 'meddelande/sänd' och 'meddelande/ström' -- SSE-strömning med terminaltillståndspridning -- Task lifecycle API:er för "tasks/get" och "tasks/cancel".
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. - -🛰️ 19. "Jag behöver riktig MCP-processhälsa, inte gissad status" +**How OmniRoute solves it:** -Operativa team måste veta om MCP faktiskt lever, inte bara om ett API är tillgängligt. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**Hur OmniRoute löser det:** + -- Runtime heartbeat-fil med PID, tidsstämplar, transport, verktygsräkning och scope-läge -- MCP status API som kombinerar hjärtslag + senaste aktivitet -- UI-statuskort för process/upptid/hjärtslagsnyhet +
+🤖 16. "I want to control model behavior globally" - -📋 20. "Jag behöver granskningsbar MCP-verktygsexekvering" +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -När verktyg muterar konfiguration eller utlöser operationsåtgärder behöver team rättsmedicinsk spårbarhet. +**How OmniRoute solves it:** -**Hur OmniRoute löser det:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- SQLite-stödd revisionsloggning för MCP-verktygsanrop -- Filtrerar efter verktyg, framgång/misslyckande, API-nyckel och paginering -- Dashboard revisionstabell + statistikslutpunkter för automatisering
+ - -🔐 21. "Jag behöver scoped MCP-behörigheter per integration" +
+🧰 17. "I need MCP tools as first-class product capabilities" -Olika klienter bör ha minst privilegierad åtkomst till verktygskategorier. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Hur OmniRoute löser det:** +**How OmniRoute solves it:** -- 10 granulära MCP-scopes för kontrollerad verktygsåtkomst -- Tillämpning av omfattning och synlighet i MCP-hanteringsgränssnitt -- Säker standardställning för operativa verktyg
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding - -⚙️ 22. "Jag behöver operativa kontroller utan att omdistribuera" + -Team behöver snabba körtidsförändringar under incidenter eller kostnadshändelser. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**Hur OmniRoute löser det:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Växla kombinationsaktivering direkt från MCP-instrumentpanelen -- Tillämpa motståndskraftsprofiler från fördefinierade policypaket -- Återställ strömbrytarens tillstånd från samma manöverpanel
+**How OmniRoute solves it:** - -🔄 23. "Jag behöver synlighet och annullering av A2A-uppgiftens livscykel" +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Utan livscykelsynlighet blir uppgiftsincidenter svåra att triage. + -**Hur OmniRoute löser det:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Uppgiftslista/filtrering efter stat/färdighet med sidnumrering -- Drill down på uppgiftens metadata, händelser och artefakter -- Slutpunkt för annullering av uppgifter och gränssnittsåtgärd med bekräftelse
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. - -🌊 24. "Jag behöver aktiv strömningsstatistik för A2A-laddning" +**How OmniRoute solves it:** -Strömmande arbetsflöden kräver operativ insikt i samtidighet och direktanslutningar. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**Hur OmniRoute löser det:** + -- Aktiva strömräknare integrerade i A2A-status -- Senaste uppgiftens tidsstämpel och antal per stat -- A2A instrumentpanelskort för operationsövervakning i realtid +
+📋 20. "I need auditable MCP tool execution" - -🪪 25. "Jag behöver standardagentupptäckt för klienter" +When tools mutate config or trigger ops actions, teams need forensic traceability. -Externa klienter och orkestratorer behöver maskinläsbar metadata för onboarding. +**How OmniRoute solves it:** -**Hur OmniRoute löser det:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- Agentkort exponerat på `/.well-known/agent.json` -- Förmåga och färdigheter som visas i ledningsgränssnittet -- A2A status API inkluderar upptäcktsmetadata för automatisering
+ - -🧭 26. "Jag behöver protokollupptäckbarhet i produktens UX" +
+🔐 21. "I need scoped MCP permissions per integration" -Om användare inte kan upptäcka protokollytor, sjunker kvaliteten på adoption och support. +Different clients should have least-privilege access to tool categories. -**Hur OmniRoute löser det:** +**How OmniRoute solves it:** -- Konsoliderad sida**Endpoints**med flikar för proxy-, MCP-, A2A- och API-slutpunkter -- Inline-tjänststatus växlar (Online/Offline) för MCP och A2A -- Länkar från översikt till dedikerade hanteringsflikar
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling - -🧪 27. "Jag behöver end-to-end protokollvalidering med riktiga klienter" + -Mock-tester räcker inte för att validera protokollkompatibilitet före release. +
+⚙️ 22. "I need operational controls without redeploying" -**Hur OmniRoute löser det:** +Teams need quick runtime changes during incidents or cost events. -- E2E-svit som startar appen och använder riktig MCP SDK-klienttransport -- A2A-klient testar för upptäckt, skicka, streama, hämta och avbryta flöden -- Korskontrollera påståenden mot MCP-revision och A2A-uppgifter API:er
+**How OmniRoute solves it:** - +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel + + + +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" + +Without lifecycle visibility, task incidents become hard to triage. + +**How OmniRoute solves it:** + +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation + +
+ +
+🌊 24. "I need active stream metrics for A2A load" + +Streaming workflows require operational insight into concurrency and live connections. + +**How OmniRoute solves it:** + +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring + +
+ +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
📡 28. "I need unified observability across all interfaces" Splitting observability by protocol creates blind spots and longer MTTR. -**Hur OmniRoute löser det:** +**How OmniRoute solves it:** - Unified dashboards/logs/analytics in one product - Health + audit + request telemetry across OpenAI, MCP, and A2A layers -- Operational APIs for status and automation
+- Operational APIs for status and automation - -💼 29. "Jag behöver en körtid för proxy + verktyg + agentorkestrering" + -Att köra många separata tjänster ökar driftskostnaderna och fellägen. +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" -**Hur OmniRoute löser det:** +Running many separate services increases operational cost and failure modes. -- OpenAI-kompatibel proxy, MCP-server och A2A-server i en stack -- Delad autentisering, resiliens, datalagring och observerbarhet -- Konsekvent policymodell över alla interaktionsytor
+**How OmniRoute solves it:** - -🚀 30. "Jag behöver skicka agentiska arbetsflöden utan limkodsprawl" +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces -Lag tappar hastighet när de sammanfogar flera ad-hoc-tjänster och skript. + -**Hur OmniRoute löser det:** +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" -- Enhetlig slutpunktsstrategi för kunder och agenter -- Inbyggda gränssnitt för protokollhantering och rökvalideringsvägar -- Produktionsfärdiga grunder (säkerhet, loggning, resiliens, backup)
+Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + + ### Example Playbooks (Integrated Use Cases) -**Playbook A: Maximera betald prenumeration + billig backup**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -607,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Playbook B: Noll-kostnad kodningsstack**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Playbook C: 24/7 alltid-på reservkedja**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -630,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Playbook D: Agent ops med MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Ställ in AI-kodning på några minuter till**$0/månad**. Anslut dessa gratiskonton och använd den inbyggda**Free Stack**-kombinationen. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Steg | Åtgärd | Leverantörer olåsta | -| ---- | ---------------------------------------------------------- | -------------------------------------------------------------------------- | -| 1 | Anslut**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**obegränsad**| -| 2 | Anslut**Qoder**(Google OAuth) | kimi-k2-tänkande, qwen3-coder-plus, deepseek-r1... —**obegränsat**| -| 3 | Anslut**Qwen**(enhetskod) | qwen3-coder-plus, qwen3-coder-flash... —**obegränsat**| -| 4 | Anslut**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180K/månad gratis**| -| 5 | `/dashboard/combos` →**Gratis stack ($0)**mall | Round-robin alla gratis leverantörer automatiskt | +| Step | Action | Providers Unlocked | +| ---- | -------------------------------------------------- | ------------------------------------------------------------------ | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Peka valfri IDE/CLI till:**`http://localhost:20128/v1` · API-nyckel: `any-string` · Klar. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Valfri extra täckning (också gratis):**Groq API-nyckel (30 rpm gratis), NVIDIA NIM (40 rpm gratis, 70+ modeller), Cerebras (1M tok/dag), LongCat API-nyckel (50M tokens/dag!), Cloudflare Workers AI (10K Neurons/day, 50+ modeller).## Snabbstart +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Snabbstart ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **pnpm-användare:**Kör `pnpm approve-builds -g` efter installationen för att aktivera inbyggda byggskript som krävs av `bättre-sqlite3` och `@swc/core`: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > > ```bash -> pnpm installera -g omniroute -> pnpm approve-builds -g # Välj alla paket → godkänn +> pnpm install -g omniroute +> pnpm approve-builds -g # Select all packages → approve > omniroute > ``` -Dashboard öppnas på `http://localhost:20128` och API-bas-URL är `http://localhost:20128/v1`. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Kommando | Beskrivning | -| ----------------------- | ----------------------------------------------------------------- | -| `omniroute` | Startserver (`PORT=20128`, API och instrumentpanel på samma port) | -| `omniroute --port 3000` | Ställ in kanonisk/API-port till 3000 | -| `omniroute --mcp` | Starta MCP-server (stdio-transport) | -| `omniroute --no-open` | Öppna inte webbläsaren automatiskt | -| `omniroute --hjälp` | Visa hjälp | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Valfritt läge med delad port:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -För de flesta distributioner behöver du bara: +For most deployments, you only need: -| Variabel | Standard | Syfte | -| ------------------------ | ------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------ | -| `REQUEST_TIMEOUT_MS` | `600000` | Delad baslinje för uppströmshämtning, dolda Undici-timeouter, TLS-fingeravtrycksbegäranden och API-bryggbegäran/proxy-timeout | -| `STREAM_IDLE_TIMEOUT_MS` | ärver `REQUEST_TIMEOUT_MS` | Maximalt gap mellan strömmande delar innan OmniRoute avbryter SSE-strömmen | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -Bakåtkompatibilitet bevaras: befintliga `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` och andra tidsgränser per lager fungerar fortfarande och åsidosätter den delade baslinjen. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Avancerade åsidosättningar är tillgängliga om du behöver bättre kontroll:| Variabel | Standard | Syfte | -| ------------------------------------------ | ------------------------------------------ | ---------------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | ärver `REQUEST_TIMEOUT_MS` | Total uppströms begäran timeout som används av huvudhämtningsavbrytsignalen | -| `FETCH_HEADERS_TIMEOUT_MS` | ärver `FETCH_TIMEOUT_MS` | Undici tidsgräns för att ta emot uppströms svarsrubriker | -| `FETCH_BODY_TIMEOUT_MS` | ärver `FETCH_TIMEOUT_MS` | Undici tidsgräns mellan uppströms kroppsdelar (`0` inaktiverar den) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30 000` | Undici TCP anslutning timeout | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | -| `TLS_CLIENT_TIMEOUT_MS` | ärver `FETCH_TIMEOUT_MS` | Timeout för TLS-fingeravtrycksbegäranden gjorda via `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | ärver `REQUEST_TIMEOUT_MS` eller `30000` | Timeout för `/v1` proxy-vidarebefordran från API-port till instrumentpanelsport | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Timeout för inkommande begäran på API-bryggservern | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60 000` | Timeout för inkommande rubrik på API-bryggservern | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout på API-bryggservern | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inaktivitet timeout på API-bryggservern (`0` inaktiverar den) | +Advanced overrides are available if you need finer control: -Om du kör OmniRoute bakom Nginx, Caddy, Cloudflare eller annan omvänd proxy, se till att proxyn -timeouts är också högre än dina OmniRoute-strömnings-/hämtningstider.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Öppna Dashboard → `Providers` och anslut minst en leverantör (OAuth- eller API-nyckel). -2. Öppna Dashboard → `Endpoints` och skapa en API-nyckel. -3. (Valfritt) Öppna Dashboard → `Combos` och ställ in din reservkedja.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Fungerar med Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode och OpenAI-kompatibla SDK:er.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (för verktygsdrivna operationer):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -Anslut sedan din MCP-klient över "stdio" och testverktyg som: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (för agent-till-agent-arbetsflöden):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -759,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Denna svit validerar riktiga MCP- och A2A-klientflöden mot en app som körs.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -767,13 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` - -Ogiltigt Linux (`xbps-src`-mall) +
+Void Linux (`xbps-src` template) -För Void Linux-användare kan du bygga ett inbyggt paket med `xbps-src`. Spara detta block som `srcpkgs/omniroute/template`:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -785,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -793,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -869,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -880,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute är tillgänglig som en offentlig Docker-bild på [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Snabbkörning:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -890,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**Med miljöfil:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Använda Docker Compose:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Dashboard-stöd för Docker-distributioner inkluderar nu en**Cloudflare Quick Tunnel**med ett klick på `Dashboard → Endpoints`. Den första aktiveringen laddar ner `cloudflared` endast när det behövs, startar en tillfällig tunnel till din nuvarande `/v1`-slutpunkt och visar den genererade `https://*.trycloudflare.com/v1`-URL:n direkt under din vanliga offentliga URL. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Anmärkningar: +Notes: -- Quick Tunnel-URL:er är tillfälliga och ändras efter varje omstart. -- Snabbtunnlar återställs inte automatiskt efter omstart av OmniRoute eller container. Återaktivera dem från instrumentpanelen vid behov. -- Hanterad installation stöder för närvarande Linux, macOS och Windows på `x64` / `arm64`. -- Managed Quick Tunnels standard till HTTP/2-transport för att undvika bullriga QUIC UDP-buffertvarningar i begränsade containermiljöer. Ställ in `CLOUDFLARED_PROTOCOL=quic` eller `auto` om du vill ha en annan transport. -- Docker-bilder buntar systemets CA-rötter och skickar dem till hanterade `cloudflared`, vilket undviker TLS-förtroendefel när tunneln startar inuti behållaren. -- SQLite körs i WAL-läge. `dockerstop` bör tillåtas avslutas så att OmniRoute kan kontrollera de senaste ändringarna tillbaka till `storage.sqlite`. -- De medföljande Compose-filerna har redan satt en frist på 40-talet. Om du kör bilden direkt, håll `--stop-timeout 40` (eller liknande) så att manuella stopp inte avbryter avstängningsrensningen. -- Ställ in `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` om du vill att OmniRoute ska använda en befintlig binär istället för att ladda ner en. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Använda Docker Compose med Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute kan exponeras säkert med Caddys automatiska SSL-provisionering. Se till att din domäns DNS A-post pekar på din servers IP.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` +| Image | Tag | Size | Description | +| ------------------------ | -------- | ------ | --------------------- | +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | -| Bild | Tagga | Storlek | Beskrivning | -| ------------------------ | -------- | ------ | ---------------------- | -| `diegosouzapw/omniroute` | `senaste` | ~250MB | Senaste stabila utgåvan | -| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Aktuell version |--- +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**NYTT!**OmniRoute är nu tillgängligt som ett**inbyggt skrivbordsprogram**för Windows, macOS och Linux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Kör OmniRoute som en fristående skrivbordsapp — ingen terminal, ingen webbläsare, inget internet krävs för lokala modeller. Den elektronbaserade appen innehåller: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Native Window**— Dedikerat appfönster med systemfältsintegration -- 🔄**Autostart**— Starta OmniRoute vid systeminloggning -- 🔔**Native Notifications**— Få varningar om kvotutmattning eller leverantörsproblem -- ⚡**One-Click Install**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Offlineläge**— Fungerar helt offline med medföljande server### Snabbstart +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Snabbstart ```bash # Development mode @@ -979,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -När den är minimerad, finns OmniRoute i systemfältet med snabba åtgärder: +When minimized, OmniRoute lives in your system tray with quick actions: -- Öppna instrumentpanelen -- Byt serverport -- Avsluta applikationen +- Open dashboard +- Change server port +- Quit application -📖 Fullständig dokumentation: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Nivå | Leverantör | Kostnad | Kvotåterställning | Bäst för | -| -------------------- | --------------------------- | -------------------------------- | ------------------------ | ----------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 PRENUMERATION** | Claude Code (Pro) | 20 USD/månad | 5h + veckovis | Har redan prenumererat | -| | Codex (Plus/Pro) | 20-200 USD/månad | 5h + veckovis | OpenAI-användare | -| | Gemini CLI | **GRATIS** | 180K/månad + 1K/dag | Alla! | -| | GitHub Copilot | $10-19/månad | Månatlig | GitHub-användare | -| **🔑 API-NYCKEL** | NVIDIA NIM | **GRATIS**(dev forever) | ~40 RPM | 70+ öppna modeller | -| | Cerebras | **GRATIS**(1M tok/dag) | 60K TPM / 30 RPM | Världens snabbaste | -| | Groq | **GRATIS**(30 RPM) | 14,4K RPD | Ultrasnabba Lama/Gemma | -| | DeepSeek V3.2 | 0,27 USD/1,10 USD per 1M | Inga | Bästa pris/kvalitetsresonemang | -| | xAI Grok-4 Fast | **$0,20/$0,50 per 1M**🆕 | Inga | Snabbast + verktygsanrop, ultralågt | -| | xAI Grok-4 (standard) | 0,20 USD/1,50 USD per 1M 🆕 | Inga | Resonerande flaggskepp från xAI | -| | Mistral | Gratis provperiod + betald | Begränsat pris | Europeisk AI | -| | OpenRouter | Betala per användning | Inga | 100+ modeller aggr. | -| **💰 BILLIGT** | GLM-5 (via Z.AI) 🆕 | $0,5/1M | Dagligen 10:00 | 128K-utgång, senaste flaggskeppet | -| | GLM-4.7 | $0,6/1M | Dagligen 10:00 | Budget backup | -| | MiniMax M2.5 🆕 | $0,3/1M ingång | 5-timmars rullande | Resonemang + agentuppgifter | -| | MiniMax M2.1 | $0,2/1M | 5-timmars rullande | Billigaste alternativet | -| | Kimi K2.5 (Moonshot API) 🆕 | Betala per användning | Inga | Direkt åtkomst till Moonshot API | -| | Kimi K2 | 9 USD/mån lägenhet | 10 miljoner tokens/månad | Förutsägbar kostnad | -| **🆓 GRATIS** | Qoder | **$0** | Obegränsad | 5 modeller obegränsat | -| | Qwen | **$0** | Obegränsad | 4 modeller obegränsat | -| | Kiro | **$0** | Obegränsad | Claude Sonnet/Haiku (AWS Builder) | -| | LongCat Flash-Lite 🆕 | **$0**(50 miljoner tok/dag 🔥) | 1 RPS | Största gratiskvoten på jorden | -| | Pollinationer AI 🆕 | **$0**(ingen nyckel behövs) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | -| | Cloudflare Workers AI 🆕 | **$0**(10 000 neuroner/dag) | ~150 resp/dag | 50+ modeller, global edge | -| | Scaleway AI 🆕 | **$0**(1 miljoner tokens totalt) | Begränsat pris | EU/GDPR, Qwen3 235B, Lama 70B | > 🆕**Nya modeller tillagda (mars 2026):**Grok-4 Fast-familjen till $0,20/$0,50/M (benchmarkerad till 1143ms — 30 % snabbare än Gemini 2.5 Flash), GLM-5 via Z.AI med 128K-utgång, MiniMax M2.5-resonemang, DeepSeek2.5-resonemang, KimSeek2.5 updates. direkt API. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 $0 Combo Stack — Den kompletta gratis installationen:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Noll kostnad. Slutar aldrig koda.**Konfigurera detta som en OmniRoute-kombination och alla fallbacks sker automatiskt – ingen manuell växling någonsin.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Alla modeller nedan är**100 % gratis utan kreditkort krävs**. OmniRoute dirigerar automatiskt mellan dem när en kvot tar slut — kombinera dem alla för en okrossbar 0 $-kombination.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Modell | Prefix | Begränsa | Prisgräns | -| ------------------ | ------ | ------------- | ---------------------- | -| `claude-sonnet-4.5` | `kr/` |**Obegränsat**| Inget rapporterat dagligt tak | -| `claude-haiku-4.5` | `kr/` |**Obegränsat**| Inget rapporterat dagligt tak | -| `claude-opus-4.6` | `kr/` |**Obegränsat**| Senaste Opus via Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) -| Modell | Prefix | Begränsa | Prisgräns | +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | --------------------- | +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | + +### 🟢 QODER MODELS (Free PAT via qodercli) + +| Model | Prefix | Limit | Rate Limit | | ------------------ | ------ | ------------- | --------------- | -| `kimi-k2-tänkande` | `om/` |**Obegränsat**| Inget rapporterat tak | -| `qwen3-coder-plus` | `om/` |**Obegränsat**| Inget rapporterat tak | -| `deepseek-r1` | `om/` |**Obegränsat**| Inget rapporterat tak | -| `minimax-m2.1` | `om/` |**Obegränsat**| Inget rapporterat tak | -| `kimi-k2` | `om/` |**Obegränsat**| Inget rapporterat tak | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -> Rekommenderad anslutningsmetod:**Personal Access Token + `qodercli`**. Webbläsarens OAuth är -> experimentell och inaktiverad som standard om inte `QODER_OAUTH_*` miljövariabler är konfigurerade.### 🟡 QWEN MODELS (Device Code Auth) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| Modell | Prefix | Begränsa | Prisgräns | -| ------------------ | ------ | ------------- | ------------------ | -| `qwen3-coder-plus` | `qw/` |**Obegränsat**| Inget rapporterat tak | -| `qwen3-coder-flash` | `qw/` |**Obegränsat**| Inget rapporterat tak | -| `qwen3-coder-next` | `qw/` |**Obegränsat**| Inget rapporterat tak | -| `vision-modell` | `qw/` |**Obegränsat**| Multimodal (bilder) |### 🟣 GEMINI CLI (Google OAuth) +### 🟡 QWEN MODELS (Device Code Auth) -| Modell | Prefix | Begränsa | Prisgräns | -| ------------------------ | ------ | -------------------------- | ------------- | -| `gemini-3-flash-preview` | `gc/` |**180K tok/månad**+ 1K/dag | Månatlig återställning | -| `gemini-2.5-pro` | `gc/` | 180K/månad (delad pool) | Hög kvalitet |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | ------------------- | +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Nivå | Daglig gräns | Prisgräns | Anteckningar | -| ---------- | ------------ | ----------- | -------------------------------------------------------------- | -| Gratis (Dev) | Ingen token cap |**~40 RPM**| 70+ modeller; övergång till rena räntegränser i mitten av 2025 | +### 🟣 GEMINI CLI (Google OAuth) -Populära gratismodeller: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-seek-instruct`/`deepseek-seek-/`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -| Nivå | Daglig gräns | Prisgräns | Anteckningar | -| ---- | ------------------ | ---------------- | -------------------------------------------------- | -| Gratis |**1M tokens/dag**| 60K TPM / 30 RPM | Världens snabbaste LLM slutledning; återställs dagligen | +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) -Tillgängligt gratis: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-destill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +| Tier | Daily Limit | Rate Limit | Notes | +| ---------- | ------------ | ----------- | ------------------------------------------------------ | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -| Nivå | Daglig gräns | Prisgräns | Anteckningar | -| ---- | ------------- | ---------------- | ------------------------------------------ | -| Gratis |**14,4K RPD**| 30 RPM per modell | Inget kreditkort; 429 på gräns, inte debiterad | +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -Tillgängligt gratis: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -| Modell | Prefix | Daglig gratiskvot | Anteckningar | -| ------------------------------ | ------ | ------------------ | ---------------------------- | -| `LongCat-Flash-Lite` | `lc/` |**50 miljoner tokens**💥 | Största gratiskvoten någonsin | -| `LongCat-Flash-Chat` | `lc/` | 500 000 tokens | Multi-turn chat | -| `LongCat-Flash-Thinking` | `lc/` | 500 000 tokens | Resonemang / CoT | -| `LongCat-Flash-Thinking-2601` | `lc/` | 500 000 tokens | Jan 2026 version | -| `LongCat-Flash-Omni-2603` | `lc/` | 500 000 tokens | Multimodal | +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -> 100 % gratis medan den är i offentlig beta. Registrera dig på [longcat.chat](https://longcat.chat) med e-post eller telefon. Återställs dagligen 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -| Modell | Prefix | Prisgräns | Provider bakom | +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ------------- | ---------------- | ----------------------------------------- | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | + +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` + +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 + +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | + +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. + +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| `openai` | `pol/` | 1 req/15s | GPT-5 | -| `claude` | `pol/` | 1 req/15s | Antropisk Claude | -| "tvillingarna" | `pol/` | 1 req/15s | Google Tvillingarna | -| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | -| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | -| `mistral` | `pol/` | 1 req/15s | Mistral AI | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**Noll friktion:**Ingen registrering, ingen API-nyckel. Lägg till pollineringsleverantören med ett tomt nyckelfält så fungerar det direkt.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| Nivå | Dagliga neuroner | Likvärdig användning | Anteckningar | -| ---- | ------------- | ----------------------------------------------- | ---------------------------- | -| Gratis |**10 000**| ~150 LLM resp / 500-tals ljud / 15K inbäddningar | Global edge, 50+ modeller | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 -Populära gratismodeller: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (gratis ljud!), `@cf/qwen/qwen2.5-coder-15b-instruct` +| Tier | Daily Neurons | Equivalent Usage | Notes | +| ---- | ------------- | --------------------------------------- | ----------------------- | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -> Kräver API-token + konto-ID från [dash.cloudflare.com](https://dash.cloudflare.com). Lagra konto-ID i leverantörsinställningar.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -| Nivå | Gratis kvot | Plats | Anteckningar | -| ---- | ------------- | ------------ | ------------------------------------------ | -| Gratis |**1M tokens**| 🇫🇷 Paris, EU | Inget kreditkort behövs inom gränserna | +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -Tillgängligt gratis: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 -> EU/GDPR-kompatibel. Hämta API-nyckel på [console.scaleway.com](https://console.scaleway.com). +| Tier | Free Quota | Location | Notes | +| ---- | ------------- | ------------ | ----------------------------------- | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | ->**💡 Den ultimata gratisstacken (11 leverantörer, $0 för alltid):** +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` + +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). + +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -> Qoder (om/) → kimi-k2-tänkande, qwen3-coder-plus, deepseek-r1 UNLIMITED -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 miljoner tokens/dag 🔥 -> Pollinationer (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — ingen nyckel behövs -> Qwen (qw/) → qwen3-kodarmodeller OBEGRÄNSAT -> Gemini (gemini/) → Gemini 2.5 Flash — 1 500 req/dag gratis -> Cloudflare AI (jfr/) → 50+ modeller — 10K neuroner/dag -> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M gratis tokens (EU) -> Groq (groq/) → Lama/Gemma — 14,4K req/dag ultrasnabb -> NVIDIA NIM (nvidia/) → 70+ öppna modeller — 40 RPM för alltid -> Cerebras (cerebras/) → Lama/Qwen världens snabbaste — 1M tok/dag -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Transkribera valfritt ljud/video för**$0**— Deepgram leads med $200 gratis, AssemblyAI $50 reserv, Groq Whisper som obegränsad nödbackup. +## 🎙️ Free Transcription Combo -| Leverantör | Gratis krediter | Bästa modellen | Prisgräns | -| ------------------ | ---------------------------- | -------------------------------------------- | ---------------------------- | -|**Deepgram**|**$200 gratis**(registrering) | `nova-3` — bästa noggrannhet, 30+ språk | Ingen RPM-gräns på gratis krediter | -| 🔵**AssemblyAI**|**$50 gratis**(registrering) | "universal-3-pro" — kapitel, sentiment, PII | Ingen RPM-gräns på gratis krediter | -| 🔴**Groq**|**Fri för alltid**| `whisper-large-v3` — OpenAI Whisper | 30 RPM (hastighetsbegränsad) | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. -**Föreslagen kombination i `/dashboard/combos`:**``` +| Provider | Free Credits | Best Model | Rate Limit | +| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | + +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Sedan i `/dashboard/media` → fliken**Transkription**: ladda upp valfri ljud- eller videofil → välj din kombinationsslutpunkt → hämta transkription i format som stöds.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 är byggd som en operativ plattform, inte bara en reläproxy.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Funktion | Vad det gör | -| ------------------------------------- | ---------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Grok-4 Fast Family** | xAI-modeller för $0,20/$0,50/M — benchmarkerade 1143ms (30 % snabbare än Gemini 2,5 Flash) | -| 🧠**GLM-5 via Z.AI** | 128K utdatakontext, $0,5/1M — senaste flaggskeppet från GLM-familjen | -| 🔮**MiniMax M2.5** | Resonemang + agentuppgifter för $0,30/1M — betydande uppgradering från M2.1 | -| 🎯**verktyg Calling Flag per modell** | Per modell 'toolCalling: true/false' i registret — AutoCombo hoppar över modeller som inte har verktygskapacitet | -| 🌍**Flerspråkig avsiktsdetektering** | PT/ZH/ES/AR-nyckelord i AutoCombo-poängsättning — bättre modellval för icke-engelsk innehåll | -| 📊**Benchmarkdrivna fallbacks** | Verklig p95 latens från liveförfrågningar feeds combo scoring — AutoCombo lär sig av faktiska data | -| 🔁**Begär deduplicering** | Innehållshashbaserat dedup-fönster – säker för flera agenter, förhindrar dubbletter av avgifter | -| 🔌**Strategi för pluggbar router** | Utökningsbart `RouterStrategy`-gränssnitt — lägg till anpassad routinglogik som plugins | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Funktion | Vad det gör | -| ------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Modell lekplats** | Dashboard-sida för att testa alla modeller direkt — leverantör/modell/slutpunktsväljare, Monaco Editor, streaming, avbryt, timing | -| 🔏**CLI Fingerprint Matching** | Beställning av rubrik/kropp per leverantör för att matcha inbyggda CLI-signaturer – växla per leverantör i Inställningar > Säkerhet.**Din proxy-IP bevaras** | -| 🤝**ACP Support (Agent Client Protocol)** | CLI-agentupptäckt (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 till), process spawner, `/api/acp/agents` slutpunkt | -| 🤖**ACP Agents Dashboard** | Debug › Agentsida — rutnät med 14 agenter med installationsstatus, version, anpassad agentform för alla CLI-verktyg.**OpenCode**-användare får en "Ladda ner opencode.json"-knapp som automatiskt genererar en färdig att använda konfiguration med alla tillgängliga modeller. | -| 🔧**Anpassad modell `apiFormat` Routing** | Anpassade modeller med `apiFormat: "responses"` dirigeras nu korrekt till Responses API-översättaren | -| 🏢**Codex Workspace Isolation** | Flera Codex-arbetsytor per e-post — OAuth separerar anslutningar korrekt efter arbetsyte-ID | -| 🔄**Automatisk uppdatering av elektroner** | Skrivbordsapp söker efter uppdateringar + automatisk installation vid omstart | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Funktion | Vad det gör | -| ---------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------- | -| 🔧**MCP-server (25 verktyg)** | IDE/agent-verktyg via 3 transporter: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 kärnor + 3 minne + 4 färdighetsverktyg | -| 🤝**A2A-server (JSON-RPC + SSE)** | Agent-till-agent-uppgiftskörning med synkroniserings- och streamingflöden | -| 🧭**Konsoliderade slutpunkterssida** | Hanteringssida med flikar med flikarna Endpoint Proxy, MCP, A2A och API Endpoints | -| 🎚️**Service aktivera/inaktivera växlar** | ON/OFF-omkopplare för MCP och A2A med inställningsbeständighet (standard: OFF) | -| 🛰️**MCP Runtime Heartbeat** | Verklig processstatus (pid, drifttid, hjärtslagsålder, transport, omfångsläge) | -| 📋**MCP Audit Trail** | Filtrerbara granskningsloggar med framgång/misslyckande och nyckeltillskrivning | -| 🔐**MCP Scope Enforcement** | 10 granulära scope-behörigheter för kontrollerad verktygsåtkomst | -| 📡**A2A Task Lifecycle Management** | Lista/filtrera uppgifter, inspektera händelser/artefakter, avbryt pågående uppgifter | -| 📋**Agent Card Discovery** | `/.well-known/agent.json` för automatisk upptäckt av klient | -| 🧪**Protokoll E2E testsele** | Verkliga MCP SDK + A2A-klientflöden i `test:protocols:e2e` | -| ⚙️**Driftskontroller** | Switch combo, applicera elasticitetsprofiler, återställ brytare från en kontrollyta | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Funktion | Vad det gör | -| ---------------------------------- | --------------------------------------------------------------------------------- | ----------------------- | -| 🎯**Smart 4-lagers reserv** | Automatisk rutt: Prenumeration → API-nyckel → Billigt → Gratis | -| 📊**Kvotspårning i realtid** | Live token count + återställ nedräkning per leverantör | -| 🔄**Formatöversättning** | OpenAI ↔ Claude ↔ Gemini ↔ Svar med schemasäkra konverteringar | -| 👥**Multi-Account Support** | Flera konton per leverantör med intelligent urval | -| 🔄**Auto Token Refresh** | OAuth-tokens uppdateras automatiskt med försök igen | -| 🎨**Anpassade kombinationer** | 9 balanseringsstrategier + reservkedjekontroll | -| 🌐**Wildcard-router** | `provider/*` dynamisk routing | -| 🧠**Tänker på budgetkontroller** | Gränser för passthrough, auto, anpassade och adaptiva resonemang | -| 🔀**Modellalias** | Inbyggd + anpassad modellaliasing och migreringssäkerhet | -| ⚡**Bakgrundsförsämring** | Rikta lågprioriterade bakgrundsuppgifter till billigare modeller | -| 🧪**Task-Aware Smart Routing** | Välj modell automatiskt efter innehållstyp (kodning/vision/analys/sammanfattning) | -| 🔄**A2A Agent Workflows** | Deterministisk FSM-orkestrator för statistiska agentavrättningar i flera steg | -| 🔀**Adaptiv routing** | Dynamisk strategiöverstyrning baserad på tokenvolym och promptkomplexitet | -| 🎲**Provider Mångfald** | Shannon entropi poäng balanserar auto-combo trafikdistribution | -| 💬**System Prompt Injection** | Globala beteendekontroller tillämpas konsekvent | -| 📄**Responses API-kompatibilitet** | Fullständigt `/v1/responses`-stöd för Codex och avancerade agentarbetsflöden | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Funktion | Vad det gör | -| ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ---------------------------------------- | -| 🖼️**Bildgenerering** | `/v1/images/generations` med moln och lokala backends | -| 📐**Inbäddningar** | `/v1/embeddings` för sök- och RAG-pipelines | -| 🎤**Ljudtranskription** | `/v1/audio/transcriptions` — 7 leverantörer (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), automatisk språkdetektering, MP4/MP3/WAV-stöd | -| 🔊**Text-till-tal** | `/v1/audio/speech` — 10 leverantörer (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) med korrekta felmeddelanden | -| 🎬**Videogenerering** | `/v1/videos/generations` (ComfyUI + SD WebUI-arbetsflöden) | -| 🎵**Music Generation** | `/v1/music/generations` (ComfyUI-arbetsflöden) | -| 🛡️**Moderationer** | `/v1/moderations` säkerhetskontroller | -| 🔀**Omrankning** | `/v1/rerank` för relevanspoäng | -| 🔍**Webbsökning**🆕 | `/v1/search` — 5 leverantörer (Serper, Brave, Perplexity, Exa, Tavily), 6 500+ gratis/månad, auto-failover, cache | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Funktion | Vad det gör | -| -------------------------------------- | ------------------------------------------------------------------------------------------------ | -------------------------------- | -| 🔌**Cremsbrytare** | Per modell trip/återhämtning med tröskelkontroller | -| 🎯**Endpoint-Aware Models** | Anpassade modeller deklarerar stödda slutpunkter + API-format | -| 🛡️**Anti-ånflock** | Mutex + semaforskydd vid återförsök/hastighetshändelser | -| 🧠**Semantisk + Signaturcache** | Kostnads-/latensminskning med två cachelager | -| ⚡**Begär idempotens** | Dubblett skyddsfönster | -| 🔒**TLS Fingerprint Spoofing** | Webbläsarliknande TLS-fingeravtryck —**minskar botdetektering och kontoflaggning** | -| 🔏**CLI Fingerprint Matching** | Matchar inbyggda CLI-begäransignaturer —**minskar förbudsrisken samtidigt som proxy-IP bevaras** | -| 🌐**IP-filtrering** | Kontroll av tillåtelse/blockeringslista för exponerade distributioner | -| 📊**Redigerbara hastighetsgränser** | Konfigurerbara globala/leverantörsnivågränser med beständighet | -| 📉**Graciös nedbrytning** | Reserver med flerskiktskapacitet som skyddar kärn-gatewayoperationer | -| 📜**Config Audit Trail** | Diff-baserad ändringsspårning förhindrar driftdrift med enkla rollbacks | -| ⏳**Provider Health Sync** | Proaktiv övervakning av tokens utgång som utlöser varningar före auktoriseringsfel | -| 🚪**Auto-inaktivera förbjudna konton** | Funktionsbrytare tätar permanent blockerade tokenkonton automatiskt | -| 🔑**API Key Management + Scoping** | Säker nyckelutgivning/rotation och modell/leverantörskontroller | -| 👁️**Omfattad API-nyckel avslöja**🆕 | Opt-in återställning av API-nycklar via `ALLOW_API_KEY_REVEAL` | -| 🛡️**Skyddade `/modeller`** | Valfri autentisering och leverantörsdöljning för modellkatalog | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Funktion | Vad det gör | -| ------------------------------------ | ------------------------------------------------------------------ | ---------------------------- | -| 📝**Begäran + proxyloggning** | Fullständig begäran/svar och proxyloggning | -| 📉**Strömmade detaljerade loggar**🆕 | Rekonstruerar SSE-nyttolastströmmar rent in i användargränssnittet | -| 📋**Unified Logs Dashboard** | Begäran, proxy, revision och konsolvyer på en sida | -| 🔍**Begär telemetri** | p50/p95/p99 latens och spårning av begäran | -| 🏥**Hälsoinstrumentpanel** | Drifttid, brytartillstånd, lockouter, cachestatistik | -| 💰**Kostnadsspårning** | Budgetkontroller och prissättning per modell | -| 📈**Analytiska visualiseringar** | Modell-/leverantörsanvändningsinsikter och trendvyer | -| 🧪**Utvärderingsram** | Golden set-testning med konfigurerbara matchstrategier | -| 📡**Live Diagnostics**🆕 | Semantisk cache-bypass för exakt combo live-testning | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Funktion | Vad det gör | -| -------------------------------------- | --------------------------------------------------------------------- | --------------------- | -| 🌐**Distribuera var som helst** | Localhost, VPS, Docker, Molnmiljöer | -| 🚇**Cloudflare Tunnel**🆕 | Snabbtunnelintegrering med ett klick från instrumentpanelen | -| 🔑**API-nyckelmodellfiltrering** | Native /v1/models-svar filtreras via tilldelade bärarkontextroller | -| ⚡**Smart Cache Bypass** | Konfigurerbar TTL-heuristik och tvingade återhämtningskontroller | -| 🔄**Säkerhetskopiering/återställning** | Export/import och katastrofåterställningsflöden | -| 🧙**Onboarding Wizard** | Första körningen guidad installation | -| 🔧**CLI Tools Dashboard** | Inställning med ett klick för populära kodningsverktyg | -| 🎮**Modell lekplats** | Testa valfri leverantör/modell/slutpunkt från instrumentpanelen | -| 🔏**CLI Fingerprint Toggle** | Fingeravtrycksmatchning per leverantör i Inställningar > Säkerhet | -| 🌐**i18n (30 språk)** | Fullständig instrumentpanel + stöd för dokumentspråk med RTL-täckning | -| 🧹**Rensa alla modeller** | Rensa modelllistan med ett klick i leverantörsinformation | -| 👁️**Sidofältskontroller**🆕 | Dölj komponenter och integrationer från Utseendeinställningar | -| 📋**Utgåvamallar** | Standardiserade GitHub-mallar för buggar och funktioner | -| 📂**Anpassad datakatalog** | `DATA_DIR` åsidosättande för lagringsplats | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1292,103 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -När kvot, ränta eller hälsa misslyckas, flyttar OmniRoute automatiskt till nästa kandidat utan manuell växling.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A kan upptäckas i användargränssnitt och dokument (inte dolda) -- Protokollstatus-API:er exponerar live operationsdata (`/api/mcp/*`, `/api/a2a/*`) -- Dashboards inkluderar åtgärder för dag-2 operationer (kombinationsväxlingar, återställning av brytare, avbokning av uppgifter)#### Translator + validation workflow +#### Protocol management that is visible and operable -Översättarområdet inkluderar: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Lekplats**: begär omvandlingskontroller -**Chatttestare**: fullständig begäran/svar tur och retur -**Testbänk**: flera fall i en körning -**Live Monitor**: trafikvy i realtid +#### Translator + validation workflow -Plus protokollvalidering med riktiga klienter via `npm run test:protocols:e2e`. +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Verktygsreferens, IDE-konfigurationer och klientexempel +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A Server README](src/lib/a2a/README.md)**— Färdigheter, JSON-RPC-metoder, streaming och uppgiftslivscykel## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute inkluderar ett inbyggt utvärderingsramverk för att testa LLM-svarskvalitet mot en gyllene uppsättning. Få åtkomst till den via**Analytics → Evals**i instrumentpanelen.### Built-in Golden Set +## 🧪 Evaluations (Evals) -Det förinstallerade "OmniRoute Golden Set" innehåller testfall för: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Hälsningar, matematik, geografi, kodgenerering -- Överensstämmelse med JSON-format, översättning, generering av markdown -- Säkerhetsvägran (skadligt innehåll), räkning, boolesk logik### Evaluation Strategies +### Built-in Golden Set -| Strategi | Beskrivning | Exempel | -| ------------ | ---------------------------------------------------- | -------------------------------- | --- | -| `exakt` | Utdata måste matcha exakt | `"4"` | -| `innehåller` | Utdata måste innehålla delsträng (skiftlägeskänslig) | `"Paris"` | -| `regex` | Utdata måste matcha regexmönster | `"1.*2.*3"` | -| `anpassad` | Anpassad JS-funktion returnerar true/false | `(output) => output.length > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) - +
🧩 MCP Setup (Model Context Protocol) -Starta MCP-transport i stdio-läge:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Rekommenderat valideringsflöde: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Anslut din MCP-klient via stdio. -2. Kör `omniroute_get_health`. -3. Kör `omniroute_list_combos`. -4. Öppna `/dashboard/mcp` för att bekräfta hjärtslag, aktivitet och granskning. - -Användbara API:er för automatisering: +Useful APIs for automation: - `GET /api/mcp/status` - `GET /api/mcp/tools` - `GET /api/mcp/audit` -- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats` - -🤝 A2A-inställning (Agent2Agent) + -Upptäck agenten:```bash +
+🤝 A2A Setup (Agent2Agent) + +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Skicka en uppgift:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` - -Hantera livscykel: +Manage lifecycle: - `GET /api/a2a/status` - `GET /api/a2a/tasks` - `GET /api/a2a/tasks/:id` -- `POST /api/a2a/tasks/:id/avbryt` +- `POST /api/a2a/tasks/:id/cancel` -Operativt användargränssnitt: +Operational UI: -- `/dashboard/a2a` för observerbarhet för uppgift/tillstånd/ström och rökåtgärder
+- `/dashboard/a2a` for task/state/stream observability and smoke actions - -🧪 End-to-end-protokollvalidering + -Validera båda protokollen med riktiga klienter:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Detta verifierar: +This verifies: -- MCP SDK-klientanslut/lista/samtal -- A2A upptäckt/skicka/strömma/få/avbryt -- Korskontrollera data i MCP-revision och A2A-uppgiftshanterings-API:er
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs - -💳 Prenumerationsleverantörer### Claude Code (Pro/Max) + + +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1401,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Proffstips:**Använd Opus för komplexa uppgifter, Sonnet för snabbhet. OmniRoute spårar kvot per modell!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1415,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Varje Codex-konto har nu policyväxlingar i `Dashboard -> Providers`: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (PÅ/AV): tillämpa policyn för 5-timmars fönstertröskel. -- "Veckovis" (PÅ/AV): tillämpa policyn för tröskelvärden för veckofönster. -- Tröskelbeteende: när ett aktiverat fönster når >=90 % användning hoppas kontot över. -- Rotationsbeteende: OmniRoute leder automatiskt till nästa kvalificerade Codex-konto. -- Återställbeteende: när leverantörens "resetAt"-tiden går, blir kontot automatiskt kvalificerat igen. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Scenarier: +Scenarios: -- `5h ON` + `Weekly ON`: kontot hoppas över när endera fönstret når tröskeln. -- `5h OFF` + `Weekly ON`: endast veckovis användning kan blockera kontot. -- `5h ON` + `Weekly OFF`: endast 5-timmars användning kan blockera kontot. -- `resetAt` godkänd: kontot återgår till rotation automatiskt (ingen manuell återaktivering).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1440,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Bäst värde:**Enorma gratis nivå! Använd detta före betalda nivåer.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1455,71 +1662,91 @@ Models:
- -🔑 API-nyckelleverantörer### NVIDIA NIM (FREE developer access — 70+ models) +
+🔑 API Key Providers -1. Registrera dig: [build.nvidia.com](https://build.nvidia.com) -2. Få gratis API-nyckel (1000 slutsatspoäng ingår) -3. Dashboard → Lägg till leverantör → NVIDIA NIM: - - API-nyckel: `nvapi-din-nyckel` +### NVIDIA NIM (FREE developer access — 70+ models) -**Modeller:**"nvidia/llama-3.3-70b-instruct", "nvidia/mistral-7b-instruct" och 50+ till +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Proffstips:**OpenAI-kompatibelt API — fungerar sömlöst med OmniRoutes formatöversättning!### DeepSeek +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -1. Registrera dig: [platform.deepseek.com](https://platform.deepseek.com) -2. Hämta API-nyckel -3. Dashboard → Lägg till leverantör → DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -**Modeller:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +### DeepSeek -1. Registrera dig: [console.groq.com](https://console.groq.com) -2. Skaffa API-nyckel (gratis nivå ingår) -3. Dashboard → Lägg till leverantör → Groq +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -**Modeller:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Proffstips:**Ultrasnabb slutledning — bäst för realtidskodning!### OpenRouter (100+ Models) +### Groq (Free Tier Available!) -1. Registrera dig: [openrouter.ai](https://openrouter.ai) -2. Hämta API-nyckel -3. Dashboard → Lägg till leverantör → OpenRouter +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -**Modeller:**Få tillgång till 100+ modeller från alla större leverantörer genom en enda API-nyckel. +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Beteende på instrumentpanelen:**OpenRouter-modeller hanteras från**Tillgängliga modeller**. Manuell tillägg, import och automatisk synkronisering uppdaterar alla samma lista.
+**Pro Tip:** Ultra-fast inference — best for real-time coding! - -💰 Billiga leverantörer (backup)### GLM-4.7 (Daily reset, $0.6/1M) +### OpenRouter (100+ Models) -1. Registrera dig: [Zhipu AI](https://open.bigmodel.cn/) -2. Hämta API-nyckel från Coding Plan -3. Instrumentpanel → Lägg till API-nyckel: - - Leverantör: `glm` - - API-nyckel: `din nyckel` +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -**Använd:**`glm/glm-4.7` +**Models:** Access 100+ models from all major providers through a single API key. -**Proffstips:**Coding Plan erbjuder 3× kvot till 1/7 kostnad! Återställ dagligen 10:00.### MiniMax M2.1 (5h reset, $0.20/1M) +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -1. Registrera dig: [MiniMax](https://www.minimax.io/) -2. Hämta API-nyckel -3. Instrumentpanel → Lägg till API-nyckel + -**Använd:**`minimax/MiniMax-M2.1` +
+💰 Cheap Providers (Backup) -**Proffstips:**Billigaste alternativet för långa sammanhang (1M tokens)!### Kimi K2 ($9/month flat) +### GLM-4.7 (Daily reset, $0.6/1M) -1. Prenumerera: [Moonshot AI](https://platform.moonshot.ai/) -2. Hämta API-nyckel -3. Instrumentpanel → Lägg till API-nyckel +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**Använd:**`kimi/kimi-latest` +**Use:** `glm/glm-4.7` -**Proffstips:**Fast $9/månad för 10 miljoner tokens = $0,90/1 miljon effektiv kostnad!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. - -🆓 GRATIS leverantörer (nödbackup)### Qoder (5 FREE models via OAuth) +### MiniMax M2.1 (5h reset, $0.20/1M) + +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `minimax/MiniMax-M2.1` + +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1560,8 +1787,10 @@ Models:
- -🎨 Skapa kombinationer### Example 1: Maximize Subscription → Cheap Backup +
+🎨 Create Combos + +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1589,8 +1818,10 @@ Cost: $0 forever!
- -🔧 CLI-integration### Cursor IDE +
+🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1601,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Använd sidan**CLI Tools**i instrumentpanelen för konfiguration med ett klick, eller redigera `~/.claude/settings.json` manuellt.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1612,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Alternativ 1 — Instrumentpanel (rekommenderas):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Alternativ 2 — Manuell:**Redigera `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1629,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Obs!**OpenClaw fungerar endast med lokala OmniRoute. Använd `127.0.0.1` istället för `localhost` för att undvika problem med IPv6-upplösning.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1643,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Steg 1:**Lägg till OmniRoute som en anpassad leverantör:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Steg 2:**Skapa/redigera `opencode.json` i din projektrot:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1669,117 +1909,130 @@ opencode } } } -```` +``` -**Steg 3:**Välj modell i OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Tips:**Lägg till valfri modell som är tillgänglig i din OmniRoute `/v1/models` slutpunkt till avsnittet `modeller`. Använd formatet `provider/model-id` från din OmniRoute-instrumentpanel.
+ --- ## Felsökning - -Klicka för att expandera felsökningsguiden +
+Click to expand troubleshooting guide -**"Språkmodellen gav inga meddelanden"** +**"Language model did not provide messages"** -- Leverantörskvoten är slut → Kontrollera instrumentpanelens kvotföljare -- Lösning: Använd kombinationsalternativ eller byt till billigare nivå +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**Taxebegränsning** +**Rate limiting** -- Prenumerationskvot ute → Fallback till GLM/MiniMax -- Lägg till kombination: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**OAuth-token har löpt ut** +**OAuth token expired** -- Automatisk uppdatering av OmniRoute -- Om problemen kvarstår: Dashboard → Leverantör → Återanslut +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Höga kostnader** +**High costs** -- Kontrollera användningsstatistik i Dashboard → Kostnader -- Byt primär modell till GLM/MiniMax -- Använd gratis nivå (Gemini CLI, Qoder) för icke-kritiska uppgifter +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Dashboard/API-portar är fel** +**Dashboard/API ports are wrong** -- `PORT` är den kanoniska basporten (och API-porten som standard) -- `API_PORT` åsidosätter endast OpenAI-kompatibel API-lyssnare -- `DASHBOARD_PORT` åsidosätter endast instrumentpanelen/Next.js-lyssnaren -- Ställ in `NEXT_PUBLIC_BASE_URL` till din instrumentpanel/offentliga webbadress (för OAuth-återuppringningar) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Molnsynkroniseringsfel** +**Cloud sync errors** -- Verifiera att `BASE_URL` pekar på din körinstans -- Verifiera "CLOUD_URL" pekar till din förväntade molnslutpunkt -- Håll `NEXT_PUBLIC_*`-värdena i linje med värden på serversidan +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Första inloggningen fungerar inte** +**First login not working** -- Kontrollera `INITIAL_PASSWORD` i `.env` -- Om det inte är inställt är reservlösenordet "123456". +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Inga förfrågningsloggar** +**No request logs** -- Begäran artefakter skrivs till `DATA_DIR/call_logs/` som en JSON-fil per begäran -- Aktivera pipelinefångst från Dashboard → Loggar → Begär loggar om du behöver detaljerade nyttolaster per steg -- Ställ in `APP_LOG_TO_FILE=true` om du också vill ha applikationskonsolloggar i `logs/application/app.log` -- Justera "APP_LOG_MAX_FILE_SIZE", "APP_LOG_RETENTION_DAYS", "APP_LOG_MAX_FILES" och "CALL_LOG_MAX_ENTRIES" efter behov +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Anslutningstest visar "Invalid" för OpenAI-kompatibla leverantörer** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Många leverantörer exponerar inte en `/models` slutpunkt -- OmniRoute v1.0.6+ inkluderar reservvalidering via chattslutföranden -- Se till att baswebbadressen innehåller suffixet `/v1`### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server - + ->**⚠️ Viktigt för användare som kör OmniRoute på en VPS, Docker eller vilken fjärrserver som helst**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -**Antigravity**och**Gemini CLI**-leverantörerna använder**Google OAuth 2.0**. Google kräver att "redirect_uri" i OAuth-flödet exakt matchar en av de förregistrerade URI:erna i appens Google Cloud Console. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -OAuth-uppgifterna som är paketerade i OmniRoute är registrerade**endast för "localhost"**. När du använder OmniRoute på en fjärrserver (t.ex. `https://omniroute.myserver.com`), avvisar Google autentiseringen med:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Du måste skapa ett**OAuth 2.0 Client ID**i Google Cloud Console med din servers URI.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Öppna Google Cloud Console** +#### Step-by-step -Gå till: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Skapa ett nytt OAuth 2.0-klient-ID** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Klicka på**"+ Skapa inloggningsuppgifter"**→**"OAuth-klient-ID"** -- Applikationstyp:**"Webbapplikation"** -- Namn: allt du gillar (t.ex. "OmniRoute Remote") +**2. Create a new OAuth 2.0 Client ID** -**3. Lägg till auktoriserade omdirigerings-URI:er** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -I fältet**"Auktoriserade omdirigerings-URI:er"**lägger du till:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Ersätt `din-server.com` med din servers domän eller IP (inkludera porten om det behövs, t.ex. `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. Spara och kopiera inloggningsuppgifterna** +After creating, Google will show the **Client ID** and **Client Secret**. -När du har skapat kommer Google att visa**klient-ID**och**klienthemlighet**. +**5. Set environment variables** -**5. Ställ in miljövariabler** +In your `.env` (or Docker environment variables): -I din `.env` (eller Docker-miljövariabler):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1788,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Starta om OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Försök ansluta igen** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Dashboard → Leverantörer → Antigravity (eller Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google kommer nu att omdirigera korrekt till `https://din-server.com/callback`.--- +--- #### Temporary workaround (without custom credentials) -Om du inte vill ställa in dina egna autentiseringsuppgifter just nu kan du fortfarande använda det**manuella URL-flödet**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute öppnar Googles auktoriserings-URL -2. Efter auktorisering försöker Google omdirigera till `localhost` (som misslyckas på fjärrservern) -3.**Kopiera hela webbadressen**från webbläsarens adressfält (även om sidan inte laddas) -4. Klistra in den URL:en i fältet som visas i OmniRoute-anslutningsmodalen -5. Klicka på**"Anslut"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Detta fungerar eftersom auktoriseringskoden i URL:en är giltig oavsett om omdirigeringssidan har laddats.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. - -🇧🇷 Versão em Português#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Os provedores**Antigravity**och**Gemini CLI**usam**Google OAuth 2.0**för autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja**exatamente**uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. +
+🇧🇷 Versão em Português -Som credenciais OAuth embutidas no OmniRoute estão cadastradas**apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Você precisa criar um**OAuth 2.0 Client ID**no Google Cloud Console com a URI do seu service.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Åtkomst till Google Cloud Console** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) **2. Crie um novo OAuth 2.0 Client ID** -- Klicka på dem**"+ Skapa inloggningsuppgifter"**→**"OAuth-klient-ID"** -- Typo de aplicativo:**"Webbapplikation"** -- Namn: escolha qualquer nome (ex: `OmniRoute Remote`) +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Adicione som auktoriserade omdirigerings-URI** +**3. Adicione as Authorized Redirect URIs** -Ingen campo**"Auktoriserade omdirigerings-URIs"**, adicione:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Substitua `seu-servidor.com` pelo domínio eller IP do seu servidor (inklusive en porta se necessário, ex: `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. Spara e kopia som credenciais** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Após criar, o Google mostrará o**Client ID**e o**Client Secret**. +**5. Configure as variáveis de ambiente** -**5. Konfigurera som variáveis de ambiente** +No seu `.env` (ou nas variáveis de ambiente do Docker): -No seu `.env` (ou nas variáveis de ambiente do Docker):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1867,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reinicie o OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute - -```` +``` **7. Tente conectar novamente** -Dashboard → Leverantörer → Antigravity (ou Gemini CLI) → OAuth +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` och autenticação funcionará.--- +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. + +--- #### Workaround temporário (sem configurar credenciais próprias) -Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo**manual de URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. O OmniRoute abrirá en URL de autorização till Google +1. O OmniRoute abrirá a URL de autorização do Google 2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) -3.**Kopiera en URL komplett**da barra de endereço do seu webbläsare (mesmo que a página não carregue) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) 4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute -5. Klicka på dem**"Anslut"** +5. Clique em **"Connect"** -> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1905,64 +2171,73 @@ Se não quiser criar credenciais próprias agora, ainda é possível usar o flux ## 🛠️ Tech Stack - -Klicka för att expandera teknisk stackinformation +
+Click to expand tech stack details --**Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ stöds**inte**— inbyggda binära filer för "better-sqlite3" är inkompatibla) --**Språk**: TypeScript 5.9 —**100 % TypeScript**över `src/` och `open-sse/` (noll `alla` i kärnmoduler sedan v2.0) --**Framework**: Next.js 16 + React 19 + Tailwind CSS 4 --**Databas**: LowDB (JSON) + SQLite (domäntillstånd + proxyloggar + MCP-granskning + routingbeslut) --**Schema**: Zod (MCP-verktyg I/O-validering, API-kontrakt) --**Protokoll**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Streaming**: Serversända händelser (SSE) --**Auth**: OAuth 2.0 (PKCE) + JWT + API-nycklar + MCP Scoped Authorization --**Test**: Node.js testlöpare + Vitest (900+ tester inklusive enhet, integration, E2E) --**CI/CD**: GitHub-åtgärder (automatisk npm-publicering + Docker Hub vid release) --**Webbplats**: [omniroute.online](https://omniroute.online) --**Paket**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Resiliens**: Strömbrytare, exponentiell backoff, anti-dundrande flock, TLS-spoofing, auto-combo självläkning
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Dokumentation -| Dokument | Beskrivning | -| ------------------------------------------------------ | ---------------------------------------------------------- | -| [Användarguide](docs/USER_GUIDE.md) | Leverantörer, kombinationer, CLI-integration, distribution | -| [API-referens](docs/API_REFERENCE.md) | Alla slutpunkter med exempel | -| [MCP-server](open-sse/mcp-server/README.md) | 16 MCP-verktyg, IDE-konfigurationer, Python/TS/Go-klienter | -| [A2A-server](src/lib/a2a/README.md) | JSON-RPC 2.0-protokoll, färdigheter, streaming, uppgiftshantering | -| [Auto-Combo Engine](docs/auto-combo.md) | 6-faktor poäng, lägespaket, självläkande | -| [Felsökning](docs/TROUBLESHOOTING.md) | Vanliga problem och lösningar | -| [Arkitektur](docs/ARCHITECTURE.md) | Systemarkitektur och interna delar | -| [Bidrar](CONTRIBUTING.md) | Utvecklingsupplägg och riktlinjer | -| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0-specifikation | -| [Säkerhetspolicy](SECURITY.md) | Sårbarhetsrapportering och säkerhetsrutiner | -| [VM-distribution](docs/VM_DEPLOYMENT_GUIDE.md) | Komplett guide: VM + nginx + Cloudflare-installation | -| [Feature Gallery](docs/FEATURES.md) | Visuell visning av instrumentpanelen med skärmdumpar | -| [Releasechecklista](docs/RELEASE_CHECKLIST.md) | Steg för validering av pre-release |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute har**210+ funktioner planerade**över flera utvecklingsfaser. Här är nyckelområdena: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Kategori | Planerade funktioner | Höjdpunkter | -| ------------------------------ | ---------------- | ------------------------------------------------------------------------------------------------------ | -| 🧠**Routing & intelligens**| 25+ | Routing med lägsta latens, taggbaserad routing, kvotförhandskontroll, val av P2C-konto | -| 🔒**Säkerhet och efterlevnad**| 20+ | SSRF-härdning, cloaking av autentiseringsuppgifter, hastighetsgräns per endpoint, hanteringsnyckelomfattning | -| 📊**Observerbarhet**| 15+ | OpenTelemetry-integration, kvotövervakning i realtid, kostnadsspårning per modell | -| 🔄**Providerintegrationer**| 20+ | Dynamiskt modellregister, nedkylning av leverantörer, Codex för flera konton, Copilot-kvotanalys | -| ⚡**Prestanda**| 15+ | Dubbla cachelager, promptcache, svarscache, streaming keepalive, batch API | -| 🌐**Ekosystem**| 10+ | WebSocket API, config hot-reload, distribuerad config store, kommersiellt läge |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**OpenCode Integration**— Inbyggt leverantörsstöd för OpenCode AI-kodnings-IDE -- 🔗**TRAE Integration**— Fullständigt stöd för TRAE AI-utvecklingsramverket -- 📦**Batch API**— Asynkron batchbearbetning för bulkförfrågningar -- 🎯**Taggbaserad routing**— Ruttbegäranden baserade på anpassade taggar och metadata -- 💰**Lägsta kostnadsstrategi**— Välj automatiskt den billigaste tillgängliga leverantören +### 🔜 Coming Soon -> 📝 Fullständiga funktionsspecifikationer tillgängliga i [`docs/new-features/`](docs/new-features/) (217 detaljerade specifikationer)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1970,18 +2245,20 @@ OmniRoute har**210+ funktioner planerade**över flera utvecklingsfaser. Här är ### How to Contribute -1. Dela förvaret -2. Skapa din funktionsgren (`git checkout -b feature/amazing-feature`) -3. Bekräfta dina ändringar (`git commit -m 'Lägg till fantastisk funktion'`) -4. Tryck till grenen (`git push origin feature/amazing-feature`) -5. Öppna en Pull Request +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Se [CONTRIBUTING.md](CONTRIBUTING.md) för detaljerade riktlinjer.### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1993,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Särskilt tack till**[9router](https://github.com/decolua/9router)**av**[decolua](https://github.com/decolua)**— originalprojektet som inspirerade denna gaffel. OmniRoute bygger på den otroliga grunden med ytterligare funktioner, multimodala API:er och en fullständig TypeScript-omskrivning. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Särskilt tack till**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— den ursprungliga Go-implementeringen som inspirerade denna JavaScript-port.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Licens -MIT-licens - se [LICENS](LICENS) för detaljer.--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/sv/docs/ARCHITECTURE.md b/docs/i18n/sv/docs/ARCHITECTURE.md index b9551e53c3..85240cadb1 100644 --- a/docs/i18n/sv/docs/ARCHITECTURE.md +++ b/docs/i18n/sv/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Senast uppdaterad: 2026-03-28_## Executive Summary -OmniRoute är en lokal AI-routinggateway och instrumentpanel byggd på Next.js. -Den tillhandahåller en enda OpenAI-kompatibel slutpunkt (`/v1/*`) och dirigerar trafik över flera uppströmsleverantörer med översättning, reserv, tokenuppdatering och användningsspårning. -Kärnfunktioner: +_Last updated: 2026-03-28_ -- OpenAI-kompatibel API-yta för CLI/verktyg (28 leverantörer) -- Begäran/svar översättning över leverantörsformat -- Modellkombination fallback (multi-modell sekvens) -- Reservkonto på kontonivå (flera konto per leverantör) -- Anslutningshantering för OAuth + API-nyckelleverantör -- Inbäddningsgenerering via `/v1/embeddings` (6 leverantörer, 9 modeller) -- Bildgenerering via `/v1/images/generations` (4 leverantörer, 9 modeller) -- Tänk taggparsning (`...`) för resonemangsmodeller -- Svarssanering för strikt OpenAI SDK-kompatibilitet -- Rollnormalisering (utvecklare→system, system→användare) för kompatibilitet mellan olika leverantörer -- Strukturerad utdatakonvertering (json_schema → Gemini responseSchema) -- Lokal beständighet för leverantörer, nycklar, alias, kombinationer, inställningar, prissättning -- Användnings-/kostnadsspårning och förfrågningsloggning -- Valfri molnsynkronisering för synkronisering av flera enheter/tillstånd -- IP-tillståndslista/blockeringslista för API-åtkomstkontroll -- Tänkande budgethantering (genomföring/auto/custom/adaptiv) -- Global systemprompt injektion -- Sessionsspårning och fingeravtryck -- Förbättrad prisbegränsning per konto med leverantörsspecifika profiler -- Strömbrytarmönster för leverantörens motståndskraft -- Åskskyddande flockskydd med mutex-låsning -- Signaturbaserad cache för begärandeduplicering -- Domänlager: modelltillgänglighet, kostnadsregler, reservpolicy, lockoutpolicy -- Beständig domäntillstånd (SQLite-genomskrivningscache för reservdelar, budgetar, lockouter, strömbrytare) -- Policymotor för centraliserad förfrågningsutvärdering (lockout → budget → reserv) -- Begär telemetri med p50/p95/p99 latensaggregation -- Korrelations-ID (X-Request-Id) för spårning från början till slut -- Loggning av efterlevnadsrevision med opt-out per API-nyckel -- Utvärderingsramverk för LLM kvalitetssäkring -- Resilience UI-instrumentpanel med strömbrytarstatus i realtid -- Modulära OAuth-leverantörer (12 individuella moduler under `src/lib/oauth/providers/`) +## Executive Summary -Primär körtidsmodell: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- Next.js-apprutter under `src/app/api/*` implementerar både instrumentpanels-API:er och kompatibilitets-API:er -- En delad SSE/routing-kärna i `src/sse/*` + `open-sse/*` hanterar leverantörsexekvering, översättning, streaming, reserv och användning## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Lokal gateway körtid -- Dashboard management API:er -- Leverantörsautentisering och tokenuppdatering -- Begär översättning och SSE-streaming -- Lokal stat + användningsbeständighet -- Valfri molnsynkroniseringsorkestrering### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Implementering av molntjänster bakom `NEXT_PUBLIC_CLOUD_URL` -- Leverantör SLA/kontrollplan utanför lokal process -- Externa CLI-binärer själva (Claude CLI, Codex CLI, etc.)## Dashboard Surface (Current) +### Out of Scope -Huvudsidor under `src/app/(dashboard)/dashboard/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — snabbstart + leverantörsöversikt +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview - `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs -- `/dashboard/providers` — leverantörsanslutningar och referenser -- `/dashboard/combos` — kombinationsstrategier, mallar, regler för modelldirigering -- `/dashboard/costs` — kostnadsaggregation och prissättningssynlighet -- `/dashboard/analytics` — användningsanalyser och utvärderingar -- `/dashboard/limits` — kvot-/kurskontroller +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls - `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation -- `/dashboard/agents` — upptäckta ACP-agenter + anpassad agentregistrering -- `/dashboard/media` — bild/video/musiklekplats -- `/dashboard/search-tools` — testning av sökleverantörer och historik -- `/dashboard/health` — drifttid, strömbrytare, hastighetsgränser +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits - `/dashboard/logs` — request/proxy/audit/console logs -- `/dashboard/inställningar` — systeminställningar flikar (allmänt, routing, kombinationsstandarder, etc.) -- `/dashboard/api-manager` — API-nyckellivscykel och modellbehörigheter## High-Level System Context +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Huvudkataloger: +Main directories: -- `src/app/api/v1/*` och `src/app/api/v1beta/*` för kompatibilitets-API:er -- `src/app/api/*` för hanterings-/konfigurations-API:er -- Nästa omskrivning i `next.config.mjs` map `/v1/*` till `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Viktiga kompatibilitetsvägar: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` – inkluderar anpassade modeller med `custom: true` -- `src/app/api/v1/embeddings/route.ts` — inbäddningsgenerering (6 leverantörer) -- `src/app/api/v1/images/generations/route.ts` — bildgenerering (4+ leverantörer inkl. Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedikerad chatt per leverantör -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedikerade inbäddningar per leverantör -- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedikerade bilder per leverantör +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` - `src/app/api/v1beta/models/[...path]/route.ts` -Hanteringsdomäner: +Management domains: -- Auth/inställningar: `src/app/api/auth/*`, `src/app/api/settings/*` -- Leverantörer/anslutningar: `src/app/api/providers*` -- Leverantörsnoder: `src/app/api/provider-nodes*` -- Anpassade modeller: `src/app/api/provider-models` (GET/POST/DELETE) -- Modellkatalog: `src/app/api/models/route.ts` (GET) -- Proxykonfiguration: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- Nycklar/alias/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- Användning: `src/app/api/usage/*` +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` - Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` -- CLI-verktygshjälpare: `src/app/api/cli-tools/*` -- IP-filter: `src/app/api/settings/ip-filter` (GET/PUT) -- Tänkande budget: `src/app/api/settings/thinking-budget` (GET/PUT) -- Systemprompt: `src/app/api/settings/system-prompt` (GET/PUT) -- Sessioner: `src/app/api/sessions` (GET) -- Prisgränser: `src/app/api/rate-limits` (GET) -- Motståndskraft: `src/app/api/resilience` (GET/PATCH) — leverantörsprofiler, strömbrytare, hastighetsgränstillstånd -- Resilience reset: `src/app/api/resilience/reset` (POST) — återställningsbrytare + nedkylningar -- Cachestatistik: `src/app/api/cache/stats` (GET/DELETE) -- Modelltillgänglighet: `src/app/api/models/availability` (GET/POST) -- Telemetri: `src/app/api/telemetry/summary` (GET) +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) - Budget: `src/app/api/usage/budget` (GET/POST) -- Reservkedjor: `src/app/api/fallback/chains` (GET/POST/DELETE) -- Efterlevnadsgranskning: `src/app/api/compliance/audit-log` (GET) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) - Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- Policyer: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Policies: `src/app/api/policies` (GET/POST) -Huvudflödesmoduler: +## 2) SSE + Translation Core -- Post: `src/sse/handlers/chat.ts` -- Kärnorkestrering: `open-sse/handlers/chatCore.ts` -- Providers exekveringsadaptrar: `open-sse/executors/*` -- Formatdetektion/leverantörskonfiguration: `open-sse/services/provider.ts` +Main flow modules: + +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` - Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Konto reservlogik: `open-sse/services/accountFallback.ts` -- Översättningsregister: `open-sse/translator/index.ts` -- Strömtransformationer: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- Användningsextraktion/normalisering: `open-sse/utils/usageTracking.ts` -- Tänk taggparser: `open-sse/utils/thinkTagParser.ts` -- Inbäddningshanterare: `open-sse/handlers/embeddings.ts` -- Inbäddningsleverantörsregister: `open-sse/config/embeddingRegistry.ts` -- Hanterare för bildgenerering: `open-sse/handlers/imageGeneration.ts` -- Bildleverantörsregister: `open-sse/config/imageRegistry.ts` -- Sanering av svar: `open-sse/handlers/responseSanitizer.ts` -- Rollnormalisering: `open-sse/services/roleNormalizer.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -Tjänster (affärslogik): +Services (business logic): -- Kontoval/poängsättning: `open-sse/services/accountSelector.ts` -- Kontextlivscykelhantering: `open-sse/services/contextManager.ts` -- Genomförande av IP-filter: `open-sse/services/ipFilter.ts` -- Sessionsspårning: `open-sse/services/sessionManager.ts` -- Begär deduplicering: `open-sse/services/signatureCache.ts` -- Injektion av systemprompt: `open-sse/services/systemPrompt.ts` -- Tänkande budgethantering: `open-sse/services/thinkingBudget.ts` -- Jokertecken modell routing: `open-sse/services/wildcardRouter.ts` -- Hantering av prisgränser: `open-sse/services/rateLimitManager.ts` -- Strömbrytare: `open-sse/services/circuitBreaker.ts` +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -Domänlagermoduler: +Domain layer modules: -- Modelltillgänglighet: `src/lib/domain/modelAvailability.ts` -- Kostnadsregler/budgetar: `src/lib/domain/costRules.ts` -- Reservpolicy: `src/lib/domain/fallbackPolicy.ts` +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` - Combo resolver: `src/lib/domain/comboResolver.ts` -- Lockoutpolicy: `src/lib/domain/lockoutPolicy.ts` -- Policymotor: `src/domain/policyEngine.ts` — centraliserad lockout → budget → reservutvärdering -- Felkodskatalog: `src/lib/domain/errorCodes.ts` -- Begärans ID: `src/lib/domain/requestId.ts` -- Timeout för hämtning: `src/lib/domain/fetchTimeout.ts` -- Begär telemetri: `src/lib/domain/requestTelemetry.ts` -- Efterlevnad/revision: `src/lib/domain/compliance/index.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` - Eval runner: `src/lib/domain/evalRunner.ts` -- Domäntillståndsbeständighet: `src/lib/db/domainState.ts` — SQLite CRUD för reservkedjor, budgetar, kostnadshistorik, lockouttillstånd, strömbrytare +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -OAuth-leverantörsmoduler (12 enskilda filer under `src/lib/oauth/providers/`): +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -- Registerindex: `src/lib/oauth/providers/index.ts` -- Individuella leverantörer: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts.`, `ts.`s.`s.` -- Tunt omslag: `src/lib/oauth/providers.ts` — återexport från enskilda moduler## 3) Persistence Layer +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -Primärt tillstånd DB (SQLite): +## 3) Persistence Layer -- Core infra: `src/lib/db/core.ts` (bättre-sqlite3, migrationer, WAL) -- Återexportera fasad: `src/lib/localDb.ts` (tunt kompatibilitetslager för uppringare) -- fil: `${DATA_DIR}/storage.sqlite` (eller `$XDG_CONFIG_HOME/omniroute/storage.sqlite` när den är inställd, annars `~/.omniroute/storage.sqlite`) -- enheter (tabeller + KV-namnrymder): providerConnections, providerNodes, modelAlias, combos, apiKeys, inställningar, prissättning,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +Primary state DB (SQLite): -Användningsbeständighet: +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -- fasad: `src/lib/usageDb.ts` (dekomponerade moduler i `src/lib/usage/*`) -- SQLite-tabeller i `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` -- valfria filartefakter kvarstår för kompatibilitet/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- äldre JSON-filer migreras till SQLite genom startmigreringar när sådana finns +Usage persistence: + +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present Domain State DB (SQLite): -- `src/lib/db/domainState.ts` — CRUD-operationer för domäntillstånd -- Tabeller (skapade i `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- Genomskrivningscachemönster: i minneskartor är auktoritativa under körning; mutationer skrivs synkront till SQLite; tillståndet återställs från DB vid kallstart## 4) Auth + Security Surfaces +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces - Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- Generering/verifiering av API-nyckel: `src/shared/utils/apiKey.ts` -- Leverantörshemligheter kvarstod i "providerConnections"-poster -- Utgående proxystöd via `open-sse/utils/proxyFetch.ts` (env vars) och `open-sse/utils/networkProxy.ts` (konfigurerbar per leverantör eller global)## 5) Cloud Sync +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync - Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Periodisk uppgift: `src/shared/services/cloudSyncScheduler.ts` -- Periodisk uppgift: `src/shared/services/modelSyncScheduler.ts` -- Styr rutt: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Reservbeslut drivs av `open-sse/services/accountFallback.ts` med hjälp av statuskoder och felmeddelandeheuristik. Kombinerad routing lägger till ett extra skydd: 400-tal som omfattas av leverantörer som uppströms innehållsblock och rollvalideringsfel behandlas som modelllokala fel så att senare kombinationsmål fortfarande kan köras.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -Uppdatering under livetrafik exekveras inuti `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -503,12 +532,14 @@ erDiagram } ``` -Fysiska lagringsfiler: +Physical storage files: -- primär runtime DB: `${DATA_DIR}/storage.sqlite` -- begära loggrader: `${DATA_DIR}/log.txt` (compat/debug-artefakt) -- strukturerade samtalsnyttolastarkiv: `${DATA_DIR}/call_logs/` -- valfria felsökningssessioner för översättare/begäran: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -543,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: kompatibilitets-API:er -- `src/app/api/v1/providers/[provider]/*`: dedikerade rutter per leverantör (chatt, inbäddningar, bilder) -- `src/app/api/providers*`: leverantör CRUD, validering, testning -- `src/app/api/provider-nodes*`: anpassad kompatibel nodhantering -- `src/app/api/provider-models`: anpassad modellhantering (CRUD) -- `src/app/api/models/route.ts`: modellkatalog-API (alias + anpassade modeller) -- `src/app/api/oauth/*`: OAuth/enhetskod flyter -- `src/app/api/keys*`: lokal API-nyckellivscykel -- `src/app/api/models/alias`: aliashantering -- `src/app/api/combos*`: reservkombinationshantering -- `src/app/api/pricing`: prissättning åsidosätter för kostnadsberäkning -- `src/app/api/settings/proxy`: proxykonfiguration (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: test för utgående proxyanslutning (POST) -- `src/app/api/usage/*`: API:er för användning och loggar -- `src/app/api/sync/*` + `src/app/api/cloud/*`: molnsynkronisering och molnvända hjälpmedel -- `src/app/api/cli-tools/*`: lokala CLI-konfigurationsförfattare/checkers -- `src/app/api/settings/ip-filter`: IP-godkännandelista/blockeringslista (GET/PUT) -- `src/app/api/settings/thinking-budget`: budgetkonfig för tänkande token (GET/PUT) -- `src/app/api/settings/system-prompt`: global systemprompt (GET/PUT) -- `src/app/api/sessions`: aktiv sessionslista (GET) -- `src/app/api/rate-limits`: räntegränsstatus per konto (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: begäran parse, kombinationshantering, kontovalsloop -- `open-sse/handlers/chatCore.ts`: översättning, exekutörsutskick, försök igen/uppdatera hantering, stream setup -- `open-sse/executors/*`: leverantörsspecifikt nätverk och formatbeteende### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: översättarregister och orkestrering -- Begär översättare: `open-sse/translator/request/*` -- Svarsöversättare: `open-sse/translator/response/*` -- Formatkonstanter: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: beständig config/state och domänbeständighet på SQLite -- `src/lib/localDb.ts`: återexport av kompatibilitet för DB-moduler -- `src/lib/usageDb.ts`: användningshistorik/samtalsloggar fasad ovanpå SQLite-tabeller## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Varje leverantör har en specialiserad exekutor som utökar `BaseExecutor` (i `open-sse/executors/base.ts`), som tillhandahåller URL-byggande, rubrikkonstruktion, försök igen med exponentiell backoff, autentiseringsuppdateringskrokar och orkestreringsmetoden `execute()`. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Exekutor | Leverantör(er) | Specialhantering | -| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------ | -| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamisk URL/header-konfiguration per leverantör | -| `AntigravityExecutor` | Google Antigravity | Anpassade projekt-/sessions-ID:n, försök igen-efter analys | -| `CodexExecutor` | OpenAI Codex | Injicerar systeminstruktioner, tvingar fram resonemang | -| `CursorExecutor` | Markör IDE | ConnectRPC-protokoll, Protobuf-kodning, begäran om signering via kontrollsumma | -| `GithubExecutor` | GitHub Copilot | Copilot token uppdatering, VSCode-härmar rubriker | -| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binärt format → SSE-konvertering | -| `GeminiCLIEexekutor` | Gemini CLI | Uppdateringscykel för Google OAuth-token | +### Persistence -Alla andra leverantörer (inklusive anpassade kompatibla noder) använder "DefaultExecutor".## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Leverantör | Format | Auth | Streama | Icke-stream | Token Refresh | Användnings-API | -| ---------------- | --------------- | ---------------------- | ---------------- | ----------- | ------------- | --------------------- | ------------------------------ | -| Claude | claude | API-nyckel / OAuth | ✅ | ✅ | ✅ | ⚠️ Endast admin | -| Tvillingarna | Tvillingarna | API-nyckel / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | -| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | -| Antigravitation | antigravitation | OAuth | ✅ | ✅ | ✅ | ✅ Full kvot API | -| OpenAI | openai | API-nyckel | ✅ | ✅ | ❌ | ❌ | -| Codex | openai-svar | OAuth | ✅ tvingad | ❌ | ✅ | ✅ Prisgränser | -| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Kvotbilder | -| Markör | markören | Anpassad kontrollsumma | ✅ | ✅ | ❌ | ❌ | -| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Användningsgränser | -| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per förfrågan | -| Qoder | openai | OAuth (Grundläggande) | ✅ | ✅ | ✅ | ⚠️ Per förfrågan | -| OpenRouter | openai | API-nyckel | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | claude | API-nyckel | ✅ | ✅ | ❌ | ❌ | -| DeepSeek | openai | API-nyckel | ✅ | ✅ | ❌ | ❌ | -| Groq | openai | API-nyckel | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | openai | API-nyckel | ✅ | ✅ | ❌ | ❌ | -| Mistral | openai | API-nyckel | ✅ | ✅ | ❌ | ❌ | -| Förvirring | openai | API-nyckel | ✅ | ✅ | ❌ | ❌ | -| Tillsammans AI | openai | API-nyckel | ✅ | ✅ | ❌ | ❌ | -| Fireworks AI | openai | API-nyckel | ✅ | ✅ | ❌ | ❌ | -| Cerebras | openai | API-nyckel | ✅ | ✅ | ❌ | ❌ | -| Sammanhålla | openai | API-nyckel | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | openai | API-nyckel | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Upptäckta källformat inkluderar: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. + +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | + +All other providers (including custom compatible nodes) use the `DefaultExecutor`. + +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: - `openai` -- `openai-svar` -- `Claude` -- "tvillingarna". +- `openai-responses` +- `claude` +- `gemini` -Målformat inkluderar: +Target formats include: -- OpenAI chatt/svar +- OpenAI chat/Responses - Claude -- Gemini/Gemini-CLI/Antigravity kuvert +- Gemini/Gemini-CLI/Antigravity envelope - Kiro -- Markör +- Cursor -Översättningar använder**OpenAI som navformat**— alla konverteringar går via OpenAI som mellanliggande:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Översättningar väljs dynamiskt baserat på källnyttolastens form och leverantörens målformat. +Additional processing layers in the translation pipeline: -Ytterligare bearbetningslager i översättningspipelinen: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Responssanering**— Tar bort icke-standardiserade fält från svar i OpenAI-format (både strömmande och icke-strömmande) för att säkerställa strikt SDK-efterlevnad --**Rollnormalisering**— Konverterar `utvecklare` → `system` för icke-OpenAI-mål; slår samman `system` → `användare` för modeller som avvisar systemrollen (GLM, ERNIE) --**Think-taggextraktion**— Parsar "..."-block från innehåll till fältet "reasoning_content" --**Structured output**— Konverterar OpenAI `response_format.json_schema` till Geminis `responseMimeType` + `responseSchema`## Supported API Endpoints +## Supported API Endpoints -| Slutpunkt | Format | Handlare | -| ---------------------------------------------------------- | ------------------ | -------------------------------------------------------------------------- | -| `POST /v1/chat/kompletteringar` | OpenAI Chat | `src/sse/handlers/chat.ts` | -| `POST /v1/meddelanden` | Claude Meddelanden | Samma hanterare (automatiskt upptäckt) | -| `POST /v1/svar` | OpenAI-svar | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/inbäddningar` | OpenAI Inbäddningar | `open-sse/handlers/embeddings.ts` | -| `GET /v1/inbäddningar` | Modelllista | API-rutt | -| `POST /v1/images/generations` | OpenAI bilder | `open-sse/handlers/imageGeneration.ts` | -| `GET /v1/images/generations` | Modelllista | API-rutt | -| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedikerad per leverantör med modellvalidering | -| `POST /v1/providers/{provider}/inbäddningar` | OpenAI Inbäddningar | Dedikerad per leverantör med modellvalidering | -| `POST /v1/providers/{provider}/images/generations` | OpenAI bilder | Dedikerad per leverantör med modellvalidering | -| `POST /v1/messages/count_tokens` | Claude Token Count | API-rutt | -| `GET /v1/modeller` | OpenAI-modelllista | API-rutt (chatt + inbäddning + bild + anpassade modeller) | -| `GET /api/modeller/katalog` | Katalog | Alla modeller grupperade efter leverantör + typ | -| `POST /v1beta/models/*:streamGenerateContent` | Tvillinginfödd | API-rutt | -| `GET/PUT/DELETE /api/settings/proxy` | Proxykonfiguration | Nätverksproxykonfiguration | -| `POST /api/settings/proxy/test` | Proxyanslutning | Proxy hälsa/anslutningstest slutpunkt | -| `GET/POST/DELETE /api/provider-models` | Leverantörsmodeller | Leverantörsmodellens metadata stödjer anpassade och hanterade tillgängliga modeller |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -Bypass-hanteraren (`open-sse/utils/bypassHandler.ts`) fångar upp kända "kastningsförfrågningar" från Claude CLI – uppvärmningsping, titelextraktioner och tokenräkningar – och returnerar ett**falskt svar**utan att konsumera uppströmsleverantörstokens. Detta utlöses endast när `User-Agent` innehåller `claude-cli`.## Request Logger Pipeline +## Bypass Handler -Begäranloggaren (`open-sse/utils/requestLogger.ts`) tillhandahåller en pipeline för felsökningsloggning i 7 steg, inaktiverad som standard, aktiverad via `ENABLE_REQUEST_LOGS=true`:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Filer skrivs till `/logs//` för varje begäranssession.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- Nedkylning av leverantörskonto på övergående/hastighets-/auth-fel -- reservkonto innan begäran misslyckas -- kombimodell fallback när nuvarande modell/leverantörsväg är uttömd## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- Förkontroll och uppdatera med ett nytt försök för uppdateringsbara leverantörer -- 401/403 försök igen efter uppdateringsförsök i kärnvägen## 3) Stream Safety +## 2) Token Expiry -- frånkopplingsmedveten strömkontroller -- översättningsström med spolning i slutet av strömmen och "[KLAR]"-hantering -- användningsuppskattning fallback när leverantörens användningsmetadata saknas## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- Synkroniseringsfel dyker upp men den lokala körtiden fortsätter -- Schemaläggaren har logik som kan försöka igen, men periodisk exekvering anropar för närvarande synkronisering med ett enda försök som standard## 5) Data Integrity +## 3) Stream Safety -- SQLite-schemamigreringar och automatisk uppgraderingskrokar vid start -- äldre JSON → SQLite-migreringskompatibilitetssökväg## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Källor för synlighet vid körning: +## 4) Cloud Sync Degradation -- konsolloggar från `src/sse/utils/logger.ts` -- användningsaggregat per begäran i SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- Fyrstegs detaljerad nyttolastfångst i SQLite (`request_detail_logs`) när `settings.detailed_logs_enabled=true` -- statusloggning för textförfrågan i `log.txt` (valfritt/kompat) -- valfria djupa förfrågningar/översättningsloggar under `loggar/` när `ENABLE_REQUEST_LOGS=true` -- dashboard-användningsslutpunkter (`/api/usage/*`) för UI-konsumtion +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -Detaljerad nyttolastfångst för begäran lagrar upp till fyra JSON-nyttolaststeg per dirigerat samtal: +## 5) Data Integrity -- rå förfrågan från kunden -- översatt begäran som faktiskt skickas uppströms -- leverantörssvar rekonstruerat som JSON; strömmade svar komprimeras till den slutliga sammanfattningen plus strömmetadata -- slutligt kundsvar som returneras av OmniRoute; streamade svar lagras i samma kompakta sammanfattningsformulär## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- JWT-hemlighet (`JWT_SECRET`) säkrar verifiering/signering av cookies på instrumentpanelen -- Initial lösenordsbootstrap ('INITIAL_PASSWORD') bör uttryckligen konfigureras för förstakörning -- API-nyckel HMAC-hemlighet (`API_KEY_SECRET`) säkrar genererat lokalt API-nyckelformat -- Leverantörshemligheter (API-nycklar/tokens) finns kvar i lokal DB och bör skyddas på filsystemnivå -- Slutpunkter för molnsynkronisering är beroende av API-nyckelbehörighet + maskin-id-semantik## Environment and Runtime Matrix +## Observability and Operational Signals -Miljövariabler som används aktivt av kod: +Runtime visibility sources: + +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption + +Detailed request payload capture stores up to four JSON payload stages per routed call: + +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: - App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` -- Lagring: `DATA_DIR` -- Kompatibelt nodbeteende: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Valfri åsidosättande av lagringsbas (Linux/macOS när 'DATA_DIR' inte är inställt): 'XDG_CONFIG_HOME' -- Säkerhetshashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Loggning: `ENABLE_REQUEST_LOGS` -- Synkronisera/molnet URL: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- Utgående proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` och varianter av små bokstäver -- SOCKS5-funktionsflaggor: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- Plattforms-/runtime-hjälpare (inte app-specifik konfiguration): "APPDATA", "NODE_ENV", "PORT", "HOSTNAME"## Known Architectural Notes +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` -1. `usageDb` och `localDb` delar samma baskatalogpolicy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) med äldre filmigrering. -2. `/api/v1/route.ts` delegerar till samma enhetliga katalogbyggare som används av `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) för att undvika semantisk drift. -3. Request logger skriver fullständiga rubriker/text när den är aktiverad; behandla loggkatalogen som känslig. -4. Molnets beteende beror på korrekt `NEXT_PUBLIC_BASE_URL` och molnets slutpunkts tillgänglighet. -5. Katalogen `open-sse/` publiceras som `@omniroute/open-sse`**npm workspace-paketet**. Källkoden importerar den via `@omniroute/open-sse/...` (löses av Next.js `transpilePackages`). Filsökvägar i det här dokumentet använder fortfarande katalognamnet `open-sse/` för konsekvens. -6. Diagram i instrumentpanelen använder**Recharts**(SVG-baserad) för tillgängliga, interaktiva analysvisualiseringar (stapeldiagram för modellanvändning, leverantörsuppdelningstabeller med framgångsfrekvenser). -7. E2E-tester använder**Playwright**(`tests/e2e/`), körs via `npm run test:e2e`. Enhetstest använder**Node.js testrunner**(`tests/unit/`), körs via `npm run test:unit`. Källkoden under `src/` är**TypeScript**(`.ts`/`.tsx`); arbetsytan `open-sse/` förblir JavaScript (`.js`). -8. Inställningssidan är organiserad i 5 flikar: Säkerhet, Routing (6 globala strategier: fill-first, round-robin, p2c, slumpmässig, minst använda, kostnadsoptimerad), Resiliens (redigerbara hastighetsgränser, strömbrytare, policyer), AI (tänkande budget, systemprompt, promptcache), Advanced (proxy).## Operational Verification Checklist +## Known Architectural Notes -- Bygg från källan: `npm run build` -- Build Docker-bild: `docker build -t omniroute .` -- Starta tjänsten och verifiera: -- `GET /api/inställningar` -- `GET /api/v1/modeller` -- CLI-målbasadressen ska vara "http://:20128/v1" när "PORT=20128" +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` +- `GET /api/v1/models` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/sv/docs/FEATURES.md b/docs/i18n/sv/docs/FEATURES.md index f5a27f66d1..fc4629bb5d 100644 --- a/docs/i18n/sv/docs/FEATURES.md +++ b/docs/i18n/sv/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Visuell guide till varje avsnitt av OmniRoute-instrumentpanelen.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Hantera AI-leverantörsanslutningar: OAuth-leverantörer (Claude Code, Codex, Gemini CLI), API-nyckelleverantörer (Groq, DeepSeek, OpenRouter) och gratisleverantörer (Qoder, Qwen, Kiro). Kiro-konton inkluderar spårning av kreditsaldo – återstående krediter, total ersättning och förnyelsedatum synligt i Dashboard → Användning.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Skapa modellroutingkombinationer med 6 strategier: prioritet, viktad, round-robin, slumpmässig, minst använda och kostnadsoptimerad. Varje kombination kedjer flera modeller med automatisk reserv och inkluderar snabba mallar och beredskapskontroller.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Omfattande användningsanalys med tokenförbrukning, kostnadsberäkningar, aktivitetsvärmekartor, veckofördelningsdiagram och uppdelningar per leverantör.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Realtidsövervakning: drifttid, minne, version, latenspercentiler (p50/p95/p99), cachestatistik och leverantörs strömbrytartillstånd.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Fyra lägen för att felsöka API-översättningar:**Lekplats**(formatomvandlare),**Chatttestare**(liveförfrågningar),**Testbänk**(batchtester) och**Live Monitor**(strömning i realtid).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Testa valfri modell direkt från instrumentbrädan. Välj leverantör, modell och slutpunkt, skriv uppmaningar med Monaco Editor, strömma svar i realtid, avbryt mitt i strömmen och visa timingstatistik.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Anpassningsbara färgteman för hela instrumentpanelen. Välj mellan 7 förinställda färger (korall, blå, röd, grön, violett, orange, cyan) eller skapa ett anpassat tema genom att välja valfri hex-färg. Stöder ljus, mörk och systemläge.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Omfattande inställningspanel med flikar: +Comprehensive settings panel with tabs: --**Allmänt**— Systemlagring, säkerhetskopieringshantering (export/import databas) -**Utseende**— Temaväljare (mörkt/ljus/system), förinställningar för färgtema och anpassade färger, synlighet i hälsologgar, synlighetskontroller för sidofältsobjekt -**Säkerhet**— API-ändpunktsskydd, anpassad leverantörsblockering, IP-filtrering, sessionsinformation -**Routing**— Modellalias, försämring av bakgrundsuppgifter -**Resiliens**— Frekvensgränsbeständighet, strömbrytarinställning, automatisk inaktivering av förbjudna konton, övervakning av leverantörens utgångsdatum -**Avancerat**— Konfigurationsförbidrag, konfigurationsrevisionsspår, reservförsämringsläge![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Konfiguration med ett klick för AI-kodningsverktyg: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor och Factory Droid. Har automatisk applicering/återställning av konfiguration, anslutningsprofiler och modellmappning.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Dashboard för att upptäcka och hantera CLI-agenter. Visar ett rutnät med 14 inbyggda agenter (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) med: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Installationsstatus**— Installerad/hittad ej med versionsdetektering -**Protokollmärken**— stdio, HTTP, etc. -**Anpassade agenter**— Registrera alla CLI-verktyg via formulär (namn, binär, versionskommando, spawn args) -**CLI Fingerprint Matching**— Växla per leverantör för att matcha inbyggda CLI-begäransignaturer, vilket minskar risken för avstängning samtidigt som proxy-IP bevaras--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Generera bilder, videor och musik från instrumentpanelen. Stöder OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open och MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Loggning av förfrågningar i realtid med filtrering efter leverantör, modell, konto och API-nyckel. Visar statuskoder, tokenanvändning, latens och svarsdetaljer.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Din enhetliga API-slutpunkt med kapacitetsuppdelning: Chattavslut, svars-API, inbäddningar, bildgenerering, omrankning, ljudtranskription, text-till-tal, moderering och registrerade API-nycklar. Cloudflare Quick Tunnel-integration och molnproxystöd för fjärråtkomst.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -Skapa, omfång och återkalla API-nycklar. Varje nyckel kan begränsas till specifika modeller/leverantörer med full åtkomst eller skrivskyddad behörighet. Visuell nyckelhantering med användningsspårning.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Administrativ åtgärdsspårning med filtrering efter åtgärdstyp, aktör, mål, IP-adress och tidsstämpel. Fullständig säkerhetshändelsehistorik.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Native Electron desktop app för Windows, macOS och Linux. Kör OmniRoute som en fristående applikation med systemfältsintegration, offlinesupport, automatisk uppdatering och installation med ett klick. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Nyckelfunktioner: +Key features: -- Polling av serverberedskap (ingen tom skärm vid kallstart) -- Systembricka med porthantering -- Innehållssäkerhetspolicy -- Engångslås -- Automatisk uppdatering vid omstart -- Plattformsbetingat gränssnitt (macOS trafikljus, Windows/Linux standardtitelfält) -- Härdat Electron build-paketering — symboliska "node_modules" i det fristående paketet upptäcks och avvisas före paketering, vilket förhindrar körtidsberoende på byggmaskinen (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Se [`electron/README.md`](../electron/README.md) för fullständig dokumentation. +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/sv/docs/TROUBLESHOOTING.md b/docs/i18n/sv/docs/TROUBLESHOOTING.md index 6859e6474b..10a5c207b9 100644 --- a/docs/i18n/sv/docs/TROUBLESHOOTING.md +++ b/docs/i18n/sv/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -Vanliga problem och lösningar för OmniRoute.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Problem | Lösning | -| --------------------------------------- | --------------------------------------------------------------------------- | --- | -| Första inloggningen fungerar inte | Ställ in `INITIAL_PASSWORD` i `.env` (ingen hårdkodad standard) | -| Instrumentpanelen öppnas vid fel port | Ställ in `PORT=20128` och `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Inga förfrågningsloggar under `loggar/` | Ställ in `ENABLE_REQUEST_LOGS=true` | -| EACCES: tillstånd nekad | Ställ in `DATA_DIR=/path/to/writable/dir` för att åsidosätta `~/.omniroute` | -| Routingstrategi sparas inte | Uppdatering till v1.4.11+ (Zod-schemafix för inställningsbeständighet) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Orsak:**Leverantörskvoten är slut. +**Cause:** Provider quota exhausted. -**Åtgärda:** +**Fix:** -1. Kontrollera instrumentpanelens kvotspårare -2. Använd en kombo med reservnivåer -3. Byt till billigare/gratis nivå### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Orsak:**Prenumerationskvoten är slut. +### Rate Limiting -**Åtgärda:** +**Cause:** Subscription quota exhausted. -- Lägg till reserv: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Använd GLM/MiniMax som billig backup### OAuth Token Expired +**Fix:** -OmniRoute uppdaterar automatiskt tokens. Om problemen kvarstår: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Instrumentpanel → Leverantör → Återanslut -2. Ta bort och lägg till leverantörsanslutningen igen--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Kontrollera att `BASE_URL` pekar på din körinstans (t.ex. `http://localhost:20128`) -2. Verifiera "CLOUD_URL" pekar på din molnslutpunkt (t.ex. "https://omniroute.dev") -3. Håll `NEXT_PUBLIC_*`-värdena i linje med värden på serversidan### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Symptom:**`Oväntad token 'd'...` på molnets slutpunkt för icke-strömmande samtal. +### Cloud `stream=false` Returns 500 -**Orsak:**Uppströms returnerar SSE-nyttolast medan klienten förväntar sig JSON. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Lösning:**Använd "stream=true" för direkta molnsamtal. Lokal körtid inkluderar SSE→JSON reserv.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Skapa en ny nyckel från den lokala instrumentpanelen (`/api/keys`) -2. Kör molnsynkronisering: Aktivera moln → Synkronisera nu -3. Gamla/icke-synkroniserade nycklar kan fortfarande returnera "401" på molnet--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Kontrollera körtidsfält: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. För portabelt läge: använd bildmål "runner-cli" (buntade CLI) -3. För värdmonteringsläge: ställ in `CLI_EXTRA_PATHS` och montera host bin-katalogen som skrivskyddad -4. Om "installed=true" och "runnable=false": binärt hittades men misslyckades med hälsokontrollen### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Kontrollera användningsstatistik i Dashboard → Användning -2. Byt primärmodell till GLM/MiniMax -3. Använd gratis nivå (Gemini CLI, Qoder) för icke-kritiska uppgifter -4. Ställ in kostnadsbudgetar per API-nyckel: Dashboard → API-nycklar → Budget--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Ställ in `ENABLE_REQUEST_LOGS=true` i din `.env`-fil. Loggar visas under katalogen `logs/`.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Huvudtillstånd: `${DATA_DIR}/storage.sqlite` (leverantörer, kombinationer, alias, nycklar, inställningar) -- Användning: SQLite-tabeller i `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + valfria `${DATA_DIR}/log.txt` och `${DATA_DIR}/call_logs/` -- Begäran loggar: `/logs/...` (när `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -När en leverantörs strömbrytare är ÖPPEN, blockeras förfrågningar tills nedkylningen går ut. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Åtgärda:** +**Fix:** -1. Gå till**Dashboard → Inställningar → Resilience** -2. Kontrollera strömbrytarkortet för den berörda leverantören -3. Klicka på**Återställ alla**för att rensa alla brytare, eller vänta tills nedkylningen löper ut -4. Kontrollera att leverantören faktiskt är tillgänglig innan du återställer### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Om en leverantör upprepade gånger går in i ÖPPET läge: +### Provider keeps tripping the circuit breaker -1. Kontrollera**Dashboard → Health → Provider Health**för felmönstret -2. Gå till**Inställningar → Resiliens → Leverantörsprofiler**och höj feltröskeln -3. Kontrollera om leverantören har ändrat API-gränser eller kräver omautentisering -4. Granska latenstelemetri — hög latens kan orsaka timeoutbaserade fel--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Se till att du använder rätt prefix: `deepgram/nova-3` eller `assemblyai/bästa` -- Kontrollera att leverantören är ansluten i**Dashboard → Leverantörer**### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Kontrollera ljudformat som stöds: "mp3", "wav", "m4a", "flac", "ogg", "webm" -- Kontrollera att filstorleken ligger inom leverantörens gränser (vanligtvis < 25 MB) -- Kontrollera giltigheten av leverantörens API-nyckel i leverantörskortet--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Använd**Dashboard → Översättare**för att felsöka formatöversättningsproblem: +Use **Dashboard → Translator** to debug format translation issues: -| Läge | När ska man använda | -| ---------------- | ----------------------------------------------------------------------------------------------------- | ------------------------ | -| **Lekplats** | Jämför in-/utdataformat sida vid sida — klistra in en misslyckad begäran för att se hur den översätts | -| **Chatttestare** | Skicka livemeddelanden och inspektera hela nyttolasten för begäran/svar inklusive rubriker | -| **Testbänk** | Kör batchtester över formatkombinationer för att hitta vilka översättningar som är trasiga | -| **Live Monitor** | Se förfrågningsflödet i realtid för att fånga intermittenta översättningsproblem | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Tänketaggar visas inte**— Kontrollera om målleverantören stöder tänkande och inställningen av tänkande budget -**Verktygsanrop avbryts**— Vissa formatöversättningar kan ta bort fält som inte stöds; verifiera i Playground-läge -**Systemprompt saknas**— Claude och Gemini hanterar systemprompter på olika sätt; kontrollera översättningsutdata -**SDK returnerar rå sträng istället för objekt**— Fixat i v1.1.0: svarssanering tar nu bort icke-standardiserade fält (`x_groq`, `usage_breakdown`, etc.) som orsakar OpenAI SDK Pydantic valideringsfel -**GLM/ERNIE avvisar 'system'-roll**— Fixat i v1.1.0: rollnormaliserare slår automatiskt samman systemmeddelanden till användarmeddelanden för inkompatibla modeller -**`utvecklarrollen inte igenkänd**– Fixad i v1.1.0: konverteras automatiskt till `system` för icke-OpenAI-leverantörer -**`json_schema` fungerar inte med Gemini**— Fixat i v1.1.0: `response_format` konverteras nu till Geminis `responseMimeType` + `responseSchema`--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- Automatisk hastighetsgräns gäller endast API-nyckelleverantörer (inte OAuth/prenumeration) -- Verifiera att**Inställningar → Motståndskraft → Leverantörsprofiler**har aktiverat automatisk hastighetsgräns -- Kontrollera om leverantören returnerar "429"-statuskoder eller "Retry-After"-rubriker### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -Leverantörsprofiler stöder dessa inställningar: +### Tuning exponential backoff --**Basfördröjning**— Initial väntetid efter första fel (standard: 1 s) -**Max fördröjning**— Maximalt väntetidstak (standard: 30s) -**Multiplikator**— Hur mycket ska fördröjningen öka per på varandra följande fel (standard: 2x)### Anti-thundering herd +Provider profiles support these settings: -När många samtidiga förfrågningar träffar en hastighetsbegränsad leverantör, använder OmniRoute mutex + automatisk hastighetsbegränsning för att serialisera förfrågningar och förhindra kaskadfel. Detta är automatiskt för API-nyckelleverantörer.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Vissa OmniRoute-användare placerar gatewayen framför RAG- eller agentstackar. I de inställningarna är det vanligt att se ett konstigt mönster: OmniRoute ser frisk ut (leverantörer upp, routingprofiler ok, inga varningar om hastighetsgränser) men det slutliga svaret är fortfarande fel. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -I praktiken kommer dessa incidenter vanligtvis från RAG-rörledningen nedströms, inte från själva gatewayen. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Om du vill ha ett delat ordförråd för att beskriva dessa misslyckanden kan du använda WFGY ProblemMap, en extern MIT-licenstextresurs som definierar sexton återkommande RAG/LLM-felmönster. På hög nivå omfattar det: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- hämtningsdrift och brutna sammanhangsgränser -- tomma eller inaktuella index och vektorlager -- inbäddning kontra semantisk oöverensstämmelse -- problem med snabb montering och sammanhangsfönster -- logisk kollaps och översäkra svar -- misslyckanden i samordning av lång kedja och agenter -- multiagentminne och rolldrift -- problem med driftsättning och bootstrap-beställning +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -Tanken är enkel: +The idea is simple: -1. När du undersöker ett dåligt svar, fånga upp: - - användaruppgift och begäran - - rutt eller leverantörskombination i OmniRoute - - alla RAG-kontexter som används nedströms (hämtade dokument, verktygsanrop, etc) -2. Kartlägg incidenten till ett eller två WFGY ProblemMap-nummer (`No.1` … `No.16`). -3. Lagra numret i din egen instrumentpanel, runbook eller incidentspårare bredvid OmniRoute-loggarna. -4. Använd motsvarande WFGY-sida för att bestämma om du behöver ändra din RAG-stack, retriever eller routingstrategi. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -Fulltext och konkreta recept finns här (MIT-licens, endast text): +Full text and concrete recipes live here (MIT license, text only): [WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -Du kan ignorera det här avsnittet om du inte kör RAG eller agentpipelines bakom OmniRoute.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**GitHub-problem**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Architecture**: Se [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) för interna detaljer -**API-referens**: Se [`docs/API_REFERENCE.md`](API_REFERENCE.md) för alla slutpunkter -**Hälsa Dashboard**: Kontrollera**Dashboard → Health**för systemstatus i realtid -**Översättare**: Använd**Dashboard → Översättare**för att felsöka formatproblem +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/sv/llm.txt b/docs/i18n/sv/llm.txt new file mode 100644 index 0000000000..93ba6d2f7e --- /dev/null +++ b/docs/i18n/sv/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Svenska) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Översikt + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Säkerhet +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/th/README.md b/docs/i18n/th/README.md index 0b9f9b4458..0561d2d441 100644 --- a/docs/i18n/th/README.md +++ b/docs/i18n/th/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_พร็อกซี API สากลของคุณ — จุดสิ้นสุดเดียว ผู้ให้บริการมากกว่า 60 ราย เวลาหยุดทำงานเป็นศูนย์ ขณะนี้มี**เซิร์ฟเวอร์ MCP (25 เครื่องมือ)**,**โปรโตคอล A2A**,**ระบบหน่วยความจำ/ทักษะ**และ**แอปอิเล็กตรอนเดสก์ท็อป**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**การแชทเสร็จสิ้น • การฝัง • การสร้างภาพ • วิดีโอ • เพลง • เสียง • การจัดอันดับใหม่ •**การค้นหาเว็บ**• เซิร์ฟเวอร์ MCP • โปรโตคอล A2A • TypeScript 100%**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _พร็อกซี API สากลของคุณ — จุดสิ้ [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 เว็บไซต์](https://omniroute.online) • [🚀 เริ่มต้นอย่างรวดเร็ว](#-quick-start) • [💡 คุณสมบัติ](#-key-features) • [📖 เอกสาร](#-documentation) • [💰 ราคา](#-pricing-at-a-glance) • [💌 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**มีจำหน่ายใน:**เชค [ภาษาอังกฤษ](README.md) | 🇧🇷 [โปรตุเกส (บราซิล)](docs/i18n/pt-BR/README.md) | 🇪🇷 [ภาษาสเปน](docs/i18n/es/README.md) | 🇫🇷 [ฝรั่งเศส](docs/i18n/fr/README.md) | 🇮🇹 [อิตาลี](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇮🇷 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [เยอรมัน](docs/i18n/de/README.md) | 🇮🇹 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | เชค [Укра́ська](docs/i18n/uk-UA/README.md) | 🇷🇷 [العربية](docs/i18n/ar/README.md) | ประเทศญี่ปุ่น [日本語](docs/i18n/ja/README.md) | 🇻🇷 [Tiếng Viết](docs/i18n/vi/README.md) | 🇧🇧 [Български](docs/i18n/bg/README.md) | 🇩🇰 [เดนมาร์ก](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇮🇹 [แมกยาร์](docs/i18n/hu/README.md) | 🇮🇩 [ภาษาอินโดนีเซีย](docs/i18n/id/README.md) | 🇰🇷 [เกาหลี](docs/i18n/ko/README.md) | 🇲🇾 [ภาษามลายู](docs/i18n/ms/README.md) | 🇮🇱 [เนเธอร์แลนด์](docs/i18n/nl/README.md) | ไว้🇴 [นอร์สค์](docs/i18n/no/README.md) | 🇧🇹 [Português (โปรตุเกส)](docs/i18n/pt/README.md) | 🇷🇴 [โรมัน](docs/i18n/ro/README.md) | 🇧🇱 [โปแลนด์](docs/i18n/pl/README.md) | 🇷🇷 [สโลวีเนีย](docs/i18n/sk/README.md) | 🇲🇪 [Svenska](docs/i18n/sv/README.md) | คาสเซิล [ฟิลิปปินส์](docs/i18n/phi/README.md) | 🇮🇿 [Šeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,554 +60,629 @@ _พร็อกซี API สากลของคุณ — จุดสิ้ ## 📸 Dashboard Preview -<รายละเอียด> +
+Click to see dashboard screenshots -คลิกเพื่อดูภาพหน้าจอแดชบอร์ด +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | -| หน้า | ภาพหน้าจอ | -| ------------------- | ---------------------------------------------------- | ---------- | -| **ผู้ให้บริการ** | ![ผู้ให้บริการ](docs/screenshots/01-providers.png) | -| **คอมโบ** | ![คอมโบ](docs/screenshots/02-combos.png) | -| **การวิเคราะห์** | ![การวิเคราะห์](docs/screenshots/03-analytics.png) | -| **สุขภาพ** | ![สุขภาพ](docs/screenshots/04-health.png) | -| **นักแปล** | ![นักแปล](docs/screenshots/05-translator.png) | -| **การตั้งค่า** | ![การตั้งค่า](docs/screenshots/06-settings.png) | -| **เครื่องมือ CLI** | ![เครื่องมือ CLI](docs/screenshots/07-cli-tools.png) | -| **บันทึกการใช้งาน** | ![การใช้งาน](docs/screenshots/08-usage.png) | -| **ปลายทาง** | ![จุดสิ้นสุด](docs/screenshots/09-endpoint.png) |
| + --- ### 🤖 Free AI Provider for your favorite coding agents -_เชื่อมต่อเครื่องมือ IDE หรือ CLI ที่ขับเคลื่อนด้วย AI ผ่าน OmniRoute — เกตเวย์ API ฟรีสำหรับการเข้ารหัสไม่จำกัด_ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ -<ตาราง> - - - -OpenClaw
-โอเพนคลอว์ -

-⭐ 205K - - - -NanoBot
-นาโนบอท -

-⭐ 20.9K - - - -PicoClaw
-พิโกคลอว์ -

-⭐ 14.6K - - - -ZeroClaw
-ซีโร่คลอว์ -

-⭐ 9.9K - - - -กรงเล็บเหล็ก
-ไอรอนคลอว์ -

-⭐ 2.1K - - - - - -OpenCode
-โอเพนโค้ด -

-⭐ 106K - - - -Codex CLI
-โคเด็กซ์ CLI -

-⭐ 60.8K - - - -รหัสโคลด
-รหัสโคลด -

-⭐ 67.3K - - - -ราศีเมถุน CLI
-ราศีเมถุน CLI -

-⭐ 94.7K - - - -รหัสกิโล
-รหัสกิโล -

-⭐ 15.5K - - - + + + + + + + + + + + + + + + +
+ + OpenClaw
+ OpenClaw +

+ ⭐ 205K +
+ + NanoBot
+ NanoBot +

+ ⭐ 20.9K +
+ + PicoClaw
+ PicoClaw +

+ ⭐ 14.6K +
+ + ZeroClaw
+ ZeroClaw +

+ ⭐ 9.9K +
+ + IronClaw
+ IronClaw +

+ ⭐ 2.1K +
+ + OpenCode
+ OpenCode +

+ ⭐ 106K +
+ + Codex CLI
+ Codex CLI +

+ ⭐ 60.8K +
+ + Claude Code
+ Claude Code +

+ ⭐ 67.3K +
+ + Gemini CLI
+ Gemini CLI +

+ ⭐ 94.7K +
+ + Kilo Code
+ Kilo Code +

+ ⭐ 15.5K +
-📡 เจ้าหน้าที่ทั้งหมดเชื่อมต่อผ่าน http://localhost:20128/v1 หรือ http://cloud.omniroute.online/v1 — การกำหนดค่าเดียว โมเดลและโควต้าไม่จำกัด--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**หยุดเสียเงินและจำกัดขีดจำกัด:** +**Stop wasting money and hitting limits:** -- โควต้าการสมัครสมาชิกจะหมดอายุโดยไม่ได้ใช้ทุกเดือน -- การจำกัดอัตราจะทำให้คุณไม่สามารถเขียนโค้ดกลางคันได้ -- API ราคาแพง ($20-50/เดือนต่อผู้ให้บริการ) -- การสลับระหว่างผู้ให้บริการด้วยตนเอง +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute แก้ปัญหานี้:** +**OmniRoute solves this:** -- ✅**เพิ่มการสมัครรับข้อมูลสูงสุด**- ติดตามโควต้า ใช้ทุกบิตก่อนรีเซ็ต -- ✅**ทางเลือกสำรองอัตโนมัติ**- การสมัครสมาชิก → คีย์ API → ราคาถูก → ฟรี ไม่มีการหยุดทำงาน -- ✅**หลายบัญชี**- หมุนเวียนระหว่างบัญชีต่อผู้ให้บริการ -- ✅**สากล**- ใช้งานได้กับ Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw และเครื่องมือ CLI ใดๆ--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💌**เข้าร่วมชุมชนของเรา!**[WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — รับความช่วยเหลือ แบ่งปันเคล็ดลับ และติดตามข่าวสารล่าสุด +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**เว็บไซต์**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**ปัญหา**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [กลุ่มชุมชน](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**การมีส่วนร่วม**: ดู [CONTRIBUTING.md](CONTRIBUTING.md) เปิดประชาสัมพันธ์ หรือเลือก `ฉบับแรกที่ดี` -**โครงการดั้งเดิม**: [9router โดย decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -เมื่อเปิดปัญหา โปรดเรียกใช้คำสั่ง system-info และแนบไฟล์ที่สร้างขึ้น:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -สิ่งนี้จะสร้าง `system-info.txt` ด้วยเวอร์ชัน Node.js, เวอร์ชัน OmniRoute, รายละเอียดระบบปฏิบัติการ, เครื่องมือ CLI ที่ติดตั้ง (qoder, gemini, claude, codex, antigravity, droid ฯลฯ) สถานะ Docker/PM2 และแพ็คเกจระบบ ทุกสิ่งที่เราต้องการในการทำซ้ำปัญหาของคุณอย่างรวดเร็ว แนบไฟล์โดยตรงกับปัญหา GitHub ของคุณ--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**นักพัฒนาทุกคนที่ใช้เครื่องมือ AI ต้องเผชิญกับปัญหาเหล่านี้ทุกวัน**OmniRoute ถูกสร้างขึ้นเพื่อแก้ไขปัญหาทั้งหมด ตั้งแต่ค่าใช้จ่ายที่มากเกินไปไปจนถึงการบล็อกระดับภูมิภาค จากโฟลว์ OAuth ที่เสียหาย ไปจนถึงการทำงานของโปรโตคอลและความสามารถในการสังเกตระดับองค์กร +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. -<รายละเอียด> -💸 1. "ฉันจ่ายค่าสมัครสมาชิกราคาแพงแต่ยังคงถูกรบกวนด้วยขีดจำกัด" +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -นักพัฒนาจ่ายเงิน $20–200/เดือนสำหรับ Claude Pro, Codex Pro หรือ GitHub Copilot แม้จะจ่ายเงิน โควต้าก็มีเพดาน — การใช้งาน 5 ชม. ขีดจำกัดรายสัปดาห์ หรือขีดจำกัดอัตราต่อนาที เซสชันการเข้ารหัสกลาง ผู้ให้บริการหยุดการตอบสนอง และนักพัฒนาสูญเสียความลื่นไหลและประสิทธิภาพการทำงาน +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**OmniRoute แก้ปัญหาอย่างไร:** +**How OmniRoute solves it:** --**ทางเลือกสำรองอัจฉริยะ 4 ระดับ**— หากโควต้าการสมัครสมาชิกหมด จะเปลี่ยนเส้นทางไปยังคีย์ API → ราคาถูก → ฟรีโดยไม่มีการแทรกแซงด้วยตนเอง --**การติดตามขีดจำกัดของผู้ให้บริการ**— สแนปช็อตโควต้าแคชรีเฟรชตามกำหนดเวลาฝั่งเซิร์ฟเวอร์ (ค่าเริ่มต้น `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) พร้อมการรีเฟรชด้วยตนเองพร้อมใช้งานใน UI --**การสนับสนุนหลายบัญชี**— หลายบัญชีต่อผู้ให้บริการพร้อมการหมุนเวียนอัตโนมัติ — เมื่อบัญชีหนึ่งหมด ให้สลับไปยังบัญชีถัดไป --**คอมโบแบบกำหนดเอง**— ห่วงโซ่ทางเลือกที่ปรับแต่งได้พร้อมกลยุทธ์การปรับสมดุล 9 แบบ (ลำดับความสำคัญ ถ่วงน้ำหนัก เติมก่อน ปัดเศษ P2C สุ่ม ใช้น้อยที่สุด ปรับต้นทุนให้เหมาะสม สุ่มเข้มงวด) --**โควต้าธุรกิจ Codex**— การตรวจสอบโควต้าพื้นที่ทำงานของธุรกิจ/ทีมโดยตรงในแดชบอร์ด
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard -<รายละเอียด> -🔌 2. "ฉันจำเป็นต้องใช้ผู้ให้บริการหลายราย แต่แต่ละรายมี API ที่แตกต่างกัน" + -OpenAI ใช้รูปแบบหนึ่ง Claude (Anthropic) ใช้อีกรูปแบบหนึ่ง Gemini ยังใช้อีกรูปแบบหนึ่ง หากผู้พัฒนาต้องการทดสอบโมเดลจากผู้ให้บริการหลายรายหรือทางเลือกระหว่างผู้ให้บริการ พวกเขาจำเป็นต้องกำหนดค่า SDK ใหม่ เปลี่ยนตำแหน่งข้อมูล และจัดการกับรูปแบบที่เข้ากันไม่ได้ ผู้ให้บริการแบบกำหนดเอง (FriendLI, NIM) มีจุดสิ้นสุดโมเดลที่ไม่ได้มาตรฐาน +
+🔌 2. "I need to use multiple providers but each has a different API" -**OmniRoute แก้ปัญหาอย่างไร:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Unified Endpoint**— `http://localhost:20128/v1` เดียวทำหน้าที่เป็นพร็อกซีสำหรับผู้ให้บริการมากกว่า 60 ราย --**การแปลรูปแบบ**— อัตโนมัติและโปร่งใส: OpenAI ↔ Claude ↔ Gemini ↔ Responses API --**การฆ่าเชื้อการตอบสนอง**— ตัดช่องที่ไม่ได้มาตรฐาน (`x_groq`, `usage_breakdown`, `service_tier`) ที่ทำลาย OpenAI SDK v1.83+ --**การปรับบทบาทให้เป็นมาตรฐาน**— แปลง `ผู้พัฒนา` → `ระบบ` สำหรับผู้ให้บริการที่ไม่ใช่ OpenAI `ระบบ` → `ผู้ใช้` สำหรับ GLM/ERNIE --**คิดการแยกแท็ก**— แยกบล็อก `<คิด>` จากโมเดลอย่าง DeepSeek R1 ให้เป็น `reasoning_content` ที่เป็นมาตรฐาน --**เอาต์พุตที่มีโครงสร้างสำหรับราศีเมถุน**— `json_schema` → `responseMimeType`/`responseSchema` การแปลงอัตโนมัติ --**`ค่าเริ่มต้นสตรีม` เป็น `false`**— สอดคล้องกับข้อกำหนด OpenAI หลีกเลี่ยง SSE ที่ไม่คาดคิดใน Python/Rust/Go SDK
+**How OmniRoute solves it:** -<รายละเอียด> -🌐 3. "ผู้ให้บริการ AI ของฉันบล็อกภูมิภาค/ประเทศของฉัน" +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -ผู้ให้บริการเช่น OpenAI/Codex บล็อกการเข้าถึงจากภูมิภาคทางภูมิศาสตร์บางแห่ง ผู้ใช้ได้รับข้อผิดพลาด เช่น `unsupported_country_region_territory` ในระหว่างการเชื่อมต่อ OAuth และ API สิ่งนี้น่าหงุดหงิดเป็นพิเศษสำหรับนักพัฒนาจากประเทศกำลังพัฒนา + -**OmniRoute แก้ปัญหาอย่างไร:** +
+🌐 3. "My AI provider blocks my region/country" --**การกำหนดค่าพร็อกซี 3 ระดับ**— พร็อกซีที่กำหนดค่าได้ 3 ระดับ: ทั่วโลก (การรับส่งข้อมูลทั้งหมด) ต่อผู้ให้บริการ (ผู้ให้บริการรายเดียวเท่านั้น) และต่อการเชื่อมต่อ/คีย์ --**ป้ายพร็อกซีที่ใช้รหัสสี**— ตัวบ่งชี้ภาพ: 🟢 พร็อกซีส่วนกลาง, 🟡 พร็อกซีผู้ให้บริการ, 🔵 พร็อกซีการเชื่อมต่อ แสดง IP เสมอ --**การแลกเปลี่ยนโทเค็น OAuth ผ่านพร็อกซี**— ขั้นตอน OAuth ยังต้องผ่านพร็อกซีด้วย เพื่อแก้ไข `unsupported_country_region_territory` --**การทดสอบการเชื่อมต่อผ่านพร็อกซี**— การทดสอบการเชื่อมต่อใช้พร็อกซีที่กำหนดค่าไว้ (ไม่มีการบายพาสโดยตรงอีกต่อไป) --**รองรับ SOCKS5**— รองรับพร็อกซี SOCKS5 เต็มรูปแบบสำหรับการกำหนดเส้นทางขาออก --**การปลอมแปลงลายนิ้วมือ TLS**— ลายนิ้วมือ TLS เหมือนเบราว์เซอร์ผ่าน `wreq-js` เพื่อเลี่ยงการตรวจจับบอท --**🔏 การจับคู่ลายนิ้วมือ CLI**— เรียงลำดับส่วนหัวและฟิลด์เนื้อหาใหม่เพื่อให้ตรงกับลายเซ็นไบนารีของ CLI ดั้งเดิม ซึ่งช่วยลดความเสี่ยงในการติดธงสถานะบัญชีได้อย่างมาก IP พร็อกซีจะถูกเก็บรักษาไว้ — คุณจะได้รับทั้งการซ่อนตัว**และ**การปกปิด IP พร้อมกัน
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. -<รายละเอียด> -🆓 4. "อยากใช้ AI เขียนโค้ด แต่ไม่มีเงิน" +**How OmniRoute solves it:** -ไม่ใช่ทุกคนที่สามารถจ่าย $20–200/เดือน สำหรับการสมัครสมาชิก AI นักศึกษา นักพัฒนาจากประเทศเกิดใหม่ ผู้ที่สมัครเล่น และฟรีแลนซ์ต้องการเข้าถึงโมเดลคุณภาพโดยไม่มีค่าใช้จ่าย +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**OmniRoute แก้ปัญหาอย่างไร:** + --**Free Tier Providers ในตัว**— รองรับเนทิฟสำหรับผู้ให้บริการฟรี 100%: Qoder (5 โมเดลไม่จำกัดผ่าน OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 โมเดลไม่จำกัด: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model) Kiro (Claude + AWS Builder ID ฟรี), Gemini CLI (ฟรี 180,000 โทเค็น/เดือน) --**Ollama Cloud**— โมเดล Ollama ที่โฮสต์บนคลาวด์ที่ `api.ollama.com` พร้อมระดับ "การใช้งานระดับเบา" ฟรี ใช้คำนำหน้า `ollamacloud/` --**คอมโบฟรีเท่านั้น**— Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/เดือน โดยไม่มีการหยุดทำงาน --**NVIDIA NIM Free Access**— ~40 RPM dev-เข้าถึงฟรีตลอดกาลสำหรับโมเดลกว่า 70 รุ่นที่ build.nvidia.com (เปลี่ยนจากเครดิตเป็นการจำกัดอัตราที่แท้จริง) --**กลยุทธ์การปรับต้นทุนให้เหมาะสม**— กลยุทธ์การกำหนดเส้นทางที่จะเลือกผู้ให้บริการที่ถูกที่สุดที่มีอยู่โดยอัตโนมัติ +
+🆓 4. "I want to use AI for coding but I have no money" -<รายละเอียด> -🔒 5. "ฉันต้องปกป้องเกตเวย์ AI ของฉันจากการเข้าถึงที่ไม่ได้รับอนุญาต" +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -เมื่อเปิดเผยเกตเวย์ AI ไปยังเครือข่าย (LAN, VPS, Docker) ใครก็ตามที่มีที่อยู่จะสามารถใช้โทเค็น/โควต้าของนักพัฒนาได้ หากไม่มีการป้องกัน API ก็เสี่ยงต่อการถูกนำไปใช้ในทางที่ผิด การแทรกทันที และการละเมิด +**How OmniRoute solves it:** -**OmniRoute แก้ปัญหาอย่างไร:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**การจัดการคีย์ API**— การสร้าง การหมุนเวียน และการกำหนดขอบเขตต่อผู้ให้บริการด้วยหน้า `/dashboard/api-manager` โดยเฉพาะ --**การอนุญาตระดับโมเดล**— จำกัดคีย์ API ให้กับโมเดลเฉพาะ (`openai/*` รูปแบบไวด์การ์ด) พร้อมสลับอนุญาตทั้งหมด/จำกัด --**การป้องกันปลายทาง API**— ต้องใช้คีย์สำหรับ `/v1/models` และบล็อกผู้ให้บริการบางรายจากรายการ --**Auth Guard + การป้องกัน CSRF**— เส้นทางแดชบอร์ดทั้งหมดที่ได้รับการป้องกันด้วยมิดเดิลแวร์ 'withAuth' + โทเค็น CSRF --**ตัวจำกัดอัตรา**— การจำกัดอัตราต่อ IP ด้วยหน้าต่างที่กำหนดค่าได้ --**การกรอง IP**— รายการที่อนุญาต/รายการบล็อกสำหรับการควบคุมการเข้าถึง --**Prompt Injection Guard**— การฆ่าเชื้อจากรูปแบบการแจ้งเตือนที่เป็นอันตราย --**การเข้ารหัส AES-256-GCM**— ข้อมูลประจำตัวได้รับการเข้ารหัสเมื่อไม่ได้ใช้งาน
+ -<รายละเอียด> -🛑 6. "ผู้ให้บริการของฉันหยุดทำงานและฉันสูญเสียขั้นตอนการเขียนโค้ด" +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -ผู้ให้บริการ AI อาจไม่เสถียร ส่งกลับข้อผิดพลาด 5xx หรือถึงขีดจำกัดอัตราชั่วคราว หากผู้พัฒนาขึ้นอยู่กับผู้ให้บริการรายเดียว ผู้ให้บริการเหล่านั้นจะถูกขัดจังหวะ หากไม่มีเซอร์กิตเบรกเกอร์ การลองซ้ำหลายครั้งอาจทำให้แอปพลิเคชันเสียหายได้ +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**OmniRoute แก้ปัญหาอย่างไร:** +**How OmniRoute solves it:** --**เซอร์กิตเบรกเกอร์ต่อรุ่น**— เปิด/ปิดอัตโนมัติด้วยเกณฑ์และคูลดาวน์ที่กำหนดค่าได้ (ปิด/เปิด/เปิดครึ่ง) กำหนดขอบเขตต่อรุ่นเพื่อหลีกเลี่ยงการบล็อกแบบเรียงซ้อน --**Exponential Backoff**— ความล่าช้าในการลองใหม่อย่างต่อเนื่อง --**Anti-Thundering Herd**— Mutex + การป้องกันเซมาฟอร์จากพายุที่ลองใหม่พร้อมกัน --**Combo Fallback Chains**— หากผู้ให้บริการหลักล้มเหลว จะตกผ่านห่วงโซ่โดยอัตโนมัติโดยไม่มีการแทรกแซง --**Combo Circuit Breaker**— ปิดการใช้งานผู้ให้บริการที่ล้มเหลวภายในคอมโบเชนโดยอัตโนมัติ --**แดชบอร์ดสุขภาพ**— การตรวจสอบสถานะการออนไลน์ สถานะของเซอร์กิตเบรกเกอร์ การล็อก สถิติแคช เวลาแฝง p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest -<รายละเอียด> -🏽 7. "การกำหนดค่าเครื่องมือ AI แต่ละรายการนั้นน่าเบื่อและซ้ำซาก" + -นักพัฒนาใช้ Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... เครื่องมือแต่ละอันจำเป็นต้องมีการกำหนดค่าที่แตกต่างกัน (จุดสิ้นสุด API, คีย์, โมเดล) การกำหนดค่าใหม่เมื่อเปลี่ยนผู้ให้บริการหรือรุ่นเป็นการเสียเวลา +
+🛑 6. "My provider went down and I lost my coding flow" -**OmniRoute แก้ปัญหาอย่างไร:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**แดชบอร์ดเครื่องมือ CLI**— หน้าเฉพาะพร้อมการตั้งค่าเพียงคลิกเดียวสำหรับ Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline --**GitHub Copilot Config Generator**— สร้าง `chatLanguageModels.json` สำหรับโค้ด VS พร้อมการเลือกโมเดลจำนวนมาก --**ตัวช่วยสร้างการเริ่มต้นใช้งาน**— คำแนะนำการตั้งค่า 4 ขั้นตอนสำหรับผู้ใช้ครั้งแรก --**จุดสิ้นสุดเดียว ทุกรุ่น**— กำหนดค่า `http://localhost:20128/v1` หนึ่งครั้ง เข้าถึงผู้ให้บริการมากกว่า 60 ราย
+**How OmniRoute solves it:** -<รายละเอียด> -🔑 8. "การจัดการโทเค็น OAuth จากผู้ให้บริการหลายรายนั้นแย่มาก" +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot — ทั้งหมดใช้ OAuth 2.0 โดยมีโทเค็นที่กำลังจะหมดอายุ นักพัฒนาจำเป็นต้องตรวจสอบสิทธิ์ซ้ำอย่างต่อเนื่อง จัดการกับ `client_secret is missing`, `redirect_uri_mismatch` และความล้มเหลวบนเซิร์ฟเวอร์ระยะไกล OAuth บน LAN/VPS เป็นปัญหาอย่างยิ่ง + -**OmniRoute แก้ปัญหาอย่างไร:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**รีเฟรชโทเค็นอัตโนมัติ**— โทเค็น OAuth รีเฟรชในพื้นหลังก่อนหมดอายุ --**OAuth 2.0 (PKCE) ในตัว**— โฟลว์อัตโนมัติสำหรับ Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder --**OAuth หลายบัญชี**— หลายบัญชีต่อผู้ให้บริการผ่านการดึงโทเค็น JWT/ID --**OAuth LAN/Remote Fix**— การตรวจจับ IP ส่วนตัวสำหรับ `redirect_uri` + โหมด URL แบบกำหนดเองสำหรับเซิร์ฟเวอร์ระยะไกล --**OAuth ที่อยู่เบื้องหลัง Nginx**— ใช้ `window.location.origin` สำหรับความเข้ากันได้ของ Reverse proxy --**คู่มือ OAuth ระยะไกล**— คำแนะนำทีละขั้นตอนสำหรับข้อมูลรับรอง Google Cloud บน VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. -<รายละเอียด> -📊 9. "ฉันไม่รู้ว่าฉันใช้จ่ายไปเท่าไหร่หรือที่ไหน" +**How OmniRoute solves it:** -นักพัฒนาซอฟต์แวร์ใช้ผู้ให้บริการแบบชำระเงินหลายราย แต่ไม่มีมุมมองการใช้จ่ายแบบรวมศูนย์ ผู้ให้บริการแต่ละรายมีแดชบอร์ดการเรียกเก็บเงินของตัวเอง แต่ไม่มีข้อมูลรวม ค่าใช้จ่ายที่ไม่คาดคิดอาจกองพะเนิน +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**OmniRoute แก้ปัญหาอย่างไร:** + --**แดชบอร์ดการวิเคราะห์ต้นทุน**— การติดตามต้นทุนต่อโทเค็นและการจัดการงบประมาณต่อผู้ให้บริการ --**ขีดจำกัดงบประมาณต่อระดับ**— เพดานการใช้จ่ายต่อระดับที่ทำให้เกิดทางเลือกสำรองอัตโนมัติ --**การกำหนดค่าราคาต่อรุ่น**— ราคาที่กำหนดค่าได้ต่อรุ่น --**สถิติการใช้งานต่อคีย์ API**— จำนวนคำขอและการประทับเวลาที่ใช้ล่าสุดต่อคีย์ --**แดชบอร์ดการวิเคราะห์**— การ์ดสถิติ แผนภูมิการใช้งานโมเดล ตารางผู้ให้บริการพร้อมอัตราความสำเร็จและเวลาแฝง +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" -<รายละเอียด> -🐛 10. "ฉันไม่สามารถวินิจฉัยข้อผิดพลาดและปัญหาในการเรียก AI ได้" +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -เมื่อการโทรล้มเหลว ผู้พัฒนาจะไม่ทราบว่าเป็นการจำกัดอัตรา โทเค็นหมดอายุ รูปแบบไม่ถูกต้อง หรือข้อผิดพลาดของผู้ให้บริการ บันทึกที่แยกส่วนในเทอร์มินัลต่างๆ หากไม่สามารถสังเกตได้ การดีบักถือเป็นการลองผิดลองถูก +**How OmniRoute solves it:** -**OmniRoute แก้ปัญหาอย่างไร:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Unified Logs Dashboard**— 4 แท็บ: บันทึกคำขอ, บันทึกพร็อกซี, บันทึกการตรวจสอบ, คอนโซล --**โปรแกรมดูบันทึกคอนโซล**— โปรแกรมดูสไตล์เทอร์มินัลแบบเรียลไทม์พร้อมระดับรหัสสี เลื่อนอัตโนมัติ ค้นหา ตัวกรอง --**SQLite Proxy Logs**— บันทึกถาวรที่รอดจากการรีสตาร์ทเซิร์ฟเวอร์ --**Translator Playground**— โหมดแก้ไขข้อบกพร่อง 4 โหมด: Playground (การแปลรูปแบบ), Chat Tester (ไป-กลับ), ม้านั่งทดสอบ (เป็นกลุ่ม), Live Monitor (เรียลไทม์) --**ร้องขอการตรวจวัดทางไกล**— p50/p95/p99 latency + การติดตาม X-Request-Id --**การบันทึกตามไฟล์พร้อมการหมุนเวียน**— บันทึกแอปหมุนเวียนตามขนาด วันที่เก็บรักษา และจำนวนการเก็บถาวร สิ่งประดิษฐ์บันทึกการโทรจะหมุนเวียนตามวันที่เก็บรักษาและจำนวนไฟล์ --**รายงานข้อมูลระบบ**— `npm run system-info` สร้าง `system-info.txt` ด้วยสภาพแวดล้อมทั้งหมดของคุณ (เวอร์ชันโหนด, เวอร์ชัน OmniRoute, OS, เครื่องมือ CLI, สถานะ Docker/PM2) แนบมาเมื่อรายงานปัญหาเพื่อการคัดแยกทันที
+ -<รายละเอียด> -🏗️ 11. "การปรับใช้และการบำรุงรักษาเกตเวย์นั้นซับซ้อน" +
+📊 9. "I don't know how much I'm spending or where" -การติดตั้ง การกำหนดค่า และการบำรุงรักษาพร็อกซี AI ในสภาพแวดล้อมที่แตกต่างกัน (ภายในเครื่อง, VPS, Docker, คลาวด์) ต้องใช้แรงงานมาก ปัญหาเช่นเส้นทางฮาร์ดโค้ด, `EACCES` ในไดเร็กทอรี, ข้อขัดแย้งของพอร์ต และการสร้างข้ามแพลตฟอร์มทำให้เกิดความขัดแย้ง +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**OmniRoute แก้ปัญหาอย่างไร:** +**How OmniRoute solves it:** --**การติดตั้ง npm ทั่วโลก**— `การติดตั้ง npm -g omniroute && omniroute` — เสร็จแล้ว --**นักเทียบท่าหลายแพลตฟอร์ม**— AMD64 + ARM64 ดั้งเดิม (Apple Silicon, AWS Graviton, Raspberry Pi) --**โปรไฟล์นักเทียบท่า**— `base` (ไม่มีเครื่องมือ CLI) และ `cli` (พร้อม Claude Code, Codex, OpenClaw) --**แอป Electron Desktop**— แอปเนทีฟสำหรับ Windows/macOS/Linux พร้อมถาดระบบ เริ่มอัตโนมัติ โหมดออฟไลน์ --**โหมดแยกพอร์ต**— API และแดชบอร์ดบนพอร์ตแยกกันสำหรับสถานการณ์ขั้นสูง (พร็อกซีย้อนกลับ เครือข่ายคอนเทนเนอร์) --**Cloud Sync**— กำหนดค่าการซิงโครไนซ์ระหว่างอุปกรณ์ผ่าน Cloudflare Workers --**การสำรองข้อมูล DB**— การสำรองข้อมูล กู้คืน ส่งออก และนำเข้าการตั้งค่าทั้งหมดโดยอัตโนมัติ พร้อมด้วย `DISABLE_SQLITE_AUTO_BACKUP` สำหรับการสำรองข้อมูลที่ได้รับการจัดการจากภายนอก
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency -<รายละเอียด> -🌍 12. "อินเทอร์เฟซเป็นภาษาอังกฤษเท่านั้น และทีมของฉันไม่พูดภาษาอังกฤษ" + -ทีมในประเทศที่ไม่ได้ใช้ภาษาอังกฤษ โดยเฉพาะในละตินอเมริกา เอเชีย และยุโรป ประสบปัญหากับอินเทอร์เฟซที่ใช้ภาษาอังกฤษเท่านั้น อุปสรรคทางภาษาลดการนำไปใช้และเพิ่มข้อผิดพลาดในการกำหนดค่า +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**OmniRoute แก้ปัญหาอย่างไร:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**แดชบอร์ด i18n — 30 ภาษา**— ทั้งหมด 500+ คีย์ที่แปล รวมถึงอารบิก บัลแกเรีย เดนมาร์ก เยอรมัน สเปน ฟินแลนด์ ฝรั่งเศส ฮิบรู ฮินดี ฮังการี อินโดนีเซีย อิตาลี ญี่ปุ่น เกาหลี มาเลย์ ดัตช์ นอร์เวย์ โปแลนด์ โปรตุเกส (PT/BR) โรมาเนีย รัสเซีย สโลวาเกีย สวีเดน ไทย ยูเครน เวียดนาม จีน ฟิลิปปินส์ อังกฤษ --**รองรับ RTL**— รองรับภาษาอาหรับและฮีบรูจากขวาไปซ้าย --**README หลายภาษา**— การแปลเอกสารฉบับสมบูรณ์ 30 รายการ --**ตัวเลือกภาษา**— ไอคอนลูกโลกในส่วนหัวสำหรับการสลับแบบเรียลไทม์
+**How OmniRoute solves it:** -<รายละเอียด> -🔄 13. "ฉันต้องการมากกว่าการแชท — ฉันต้องการการฝัง รูปภาพ เสียง" +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -AI ไม่ใช่แค่การแชทให้เสร็จสิ้นเท่านั้น นักพัฒนาจำเป็นต้องสร้างภาพ ถอดเสียง สร้างการฝังสำหรับ RAG จัดอันดับเอกสารใหม่ และกลั่นกรองเนื้อหา API แต่ละรายการมีจุดสิ้นสุดและรูปแบบที่แตกต่างกัน + -**OmniRoute แก้ปัญหาอย่างไร:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**การฝัง**— `/v1/embeddings` พร้อมผู้ให้บริการ 6 รายและโมเดลมากกว่า 9 รายการ --**การสร้างภาพ**— `/v1/images/ generations` พร้อมผู้ให้บริการ 10 รายและโมเดลมากกว่า 20 รุ่น (OpenAI, xAI, Together, ดอกไม้ไฟ, Nebius, ไฮเปอร์โบลิก, NanoBanana, ต้านแรงโน้มถ่วง, SD WebUI, ComfyUI) --**การแปลงข้อความเป็นวิดีโอ**— `/v1/videos/ generations` — ComfyUI (AnimateDiff, SVD) และ SD WebUI --**การแปลงข้อความเป็นเพลง**— `/v1/music/ generations` — ComfyUI (Stable Audio Open, MusicGen) --**การถอดเสียง**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**การอ่านออกเสียงข้อความ**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + ผู้ให้บริการที่มีอยู่ --**การกลั่นกรอง**— `/v1/moderations` — การตรวจสอบความปลอดภัยของเนื้อหา --**การจัดอันดับใหม่**— `/v1/rerank` — การจัดอันดับความเกี่ยวข้องของเอกสารใหม่ --**Responses API**— รองรับ `/v1/responses` เต็มรูปแบบสำหรับ Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. -<รายละเอียด> -🧪 14. "ฉันไม่สามารถทดสอบและเปรียบเทียบคุณภาพระหว่างรุ่นต่างๆ ได้" +**How OmniRoute solves it:** -นักพัฒนาต้องการทราบว่าโมเดลใดดีที่สุดสำหรับกรณีการใช้งานของพวกเขา เช่น โค้ด การแปล การใช้เหตุผล แต่การเปรียบเทียบด้วยตนเองนั้นช้า ไม่มีเครื่องมือประเมินแบบรวมอยู่ +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**OmniRoute แก้ปัญหาอย่างไร:** + --**การประเมิน LLM**— การทดสอบชุดทองพร้อมเคสที่โหลดไว้ล่วงหน้า 10 เคส ซึ่งครอบคลุมการทักทาย คณิตศาสตร์ ภูมิศาสตร์ การสร้างโค้ด การปฏิบัติตาม JSON การแปล การมาร์กดาวน์ การปฏิเสธด้านความปลอดภัย --**4 กลยุทธ์การจับคู่**— `แน่นอน`, `มี`, `regex`, `กำหนดเอง` (ฟังก์ชัน JS) --**ม้านั่งทดสอบสนามเด็กเล่นสำหรับนักแปล**— การทดสอบเป็นกลุ่มที่มีอินพุตหลายอินพุตและเอาต์พุตที่คาดหวัง การเปรียบเทียบข้ามผู้ให้บริการ --**เครื่องมือทดสอบการแชท**— ไป-กลับเต็มรูปแบบพร้อมการเรนเดอร์การตอบสนองด้วยภาพ --**Live Monitor**— สตรีมคำขอทั้งหมดที่ไหลผ่านพร็อกซีแบบเรียลไทม์ +
+🌍 12. "The interface is English-only and my team doesn't speak English" -<รายละเอียด> -📈 15. "ฉันต้องปรับขนาดโดยไม่สูญเสียประสิทธิภาพ" +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -เมื่อปริมาณคำขอเพิ่มขึ้น หากไม่มีการแคช คำถามเดียวกันจะทำให้เกิดต้นทุนซ้ำซ้อน หากไม่มีการระบุตัวตน คำขอซ้ำจะสูญเปล่าในการประมวลผล ต้องเคารพขีดจำกัดอัตราต่อผู้ให้บริการ +**How OmniRoute solves it:** -**OmniRoute แก้ปัญหาอย่างไร:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Semantic Cache**— แคชสองชั้น (ลายเซ็น + ความหมาย) ช่วยลดต้นทุนและเวลาแฝง --**คำขอ Idempotency**— หน้าต่างการขจัดข้อมูลซ้ำซ้อน 5 วินาทีสำหรับคำขอที่เหมือนกัน --**การตรวจจับขีดจำกัดอัตรา**— RPM ต่อผู้ให้บริการ ช่องว่างขั้นต่ำ และการติดตามพร้อมกันสูงสุด --**ขีดจำกัดอัตราที่แก้ไขได้**— ค่าเริ่มต้นที่กำหนดค่าได้ในการตั้งค่า → ความยืดหยุ่นด้วยความคงอยู่ --**แคชการตรวจสอบคีย์ API**— แคช 3 ระดับสำหรับประสิทธิภาพการผลิต --**แดชบอร์ดสุขภาพพร้อมการวัดและส่งข้อมูลทางไกล**— เวลาแฝง p50/p95/p99 สถิติแคช เวลาทำงาน
+ -<รายละเอียด> -🤖 16. "ฉันต้องการควบคุมพฤติกรรมของโมเดลทั่วโลก" +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -นักพัฒนาที่ต้องการคำตอบทั้งหมดในภาษาใดภาษาหนึ่ง มีน้ำเสียงเฉพาะ หรือต้องการจำกัดโทเค็นการให้เหตุผล การกำหนดค่านี้ในทุกเครื่องมือ/คำขอไม่สามารถทำได้ +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**OmniRoute แก้ปัญหาอย่างไร:** +**How OmniRoute solves it:** --**การแจ้งพร้อมท์ของระบบ**— พร้อมท์ทั่วโลกนำไปใช้กับคำขอทั้งหมด --**การตรวจสอบงบประมาณการคิด**— การควบคุมการจัดสรรโทเค็นการให้เหตุผลต่อคำขอ (การส่งผ่าน, อัตโนมัติ, กำหนดเอง, การปรับตัว) --**9 กลยุทธ์การกำหนดเส้นทาง**— กลยุทธ์ระดับโลกที่กำหนดวิธีกระจายคำขอ --**Wildcard Router**— รูปแบบ `ผู้ให้บริการ/*` กำหนดเส้นทางแบบไดนามิกไปยังผู้ให้บริการใดๆ --**Combo Enable/Disable Toggle**— สลับคอมโบได้โดยตรงจากแดชบอร์ด --**สลับผู้ให้บริการ**— เปิด/ปิดการเชื่อมต่อทั้งหมดสำหรับผู้ให้บริการได้ด้วยคลิกเดียว --**ผู้ให้บริการที่ถูกบล็อก**— ยกเว้นผู้ให้บริการบางรายจากรายการ `/v1/models`
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex -<รายละเอียด> -🧰 17. "ฉันต้องการเครื่องมือ MCP เนื่องจากความสามารถของผลิตภัณฑ์ระดับเฟิร์สคลาส" + -เกตเวย์ AI จำนวนมากเปิดเผย MCP เป็นเพียงรายละเอียดการใช้งานที่ซ่อนอยู่เท่านั้น ทีมจำเป็นต้องมีชั้นการปฏิบัติงานที่มองเห็นได้และจัดการได้ +
+🧪 14. "I have no way to test and compare quality across models" -**OmniRoute แก้ปัญหาอย่างไร:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP ปรากฏในการนำทางแดชบอร์ดและแท็บโปรโตคอลปลายทาง -- หน้าการจัดการ MCP เฉพาะพร้อมกระบวนการ เครื่องมือ ขอบเขต และการตรวจสอบ -- การเริ่มต้นอย่างรวดเร็วในตัวสำหรับ `omniroute --mcp` และการเริ่มต้นใช้งานไคลเอ็นต์
+**How OmniRoute solves it:** -<รายละเอียด> -🧠 18. "ฉันต้องการการประสาน A2A พร้อมเส้นทางงานการซิงค์และสตรีม" +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -เวิร์กโฟลว์ของตัวแทนต้องการทั้งการตอบกลับโดยตรงและการดำเนินการสตรีมที่ใช้เวลานานพร้อมการควบคุมวงจรชีวิต + -**OmniRoute แก้ปัญหาอย่างไร:** +
+📈 15. "I need to scale without losing performance" -- จุดสิ้นสุด A2A JSON-RPC (`POST /a2a`) พร้อมด้วย `ข้อความ/ส่ง` และ `ข้อความ/สตรีม` -- การสตรีม SSE พร้อมการเผยแพร่สถานะเทอร์มินัล -- API วงจรชีวิตของงานสำหรับ `งาน/รับ` และ `งาน/ยกเลิก`
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. -<รายละเอียด> -🛰️ 19. "ฉันต้องการความสมบูรณ์ของกระบวนการ MCP ที่แท้จริง ไม่ใช่สถานะที่เดาได้" +**How OmniRoute solves it:** -ทีมปฏิบัติการจำเป็นต้องทราบว่า MCP ยังคงอยู่จริงหรือไม่ ไม่ใช่แค่ว่า API สามารถเข้าถึงได้หรือไม่ +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**OmniRoute แก้ปัญหาอย่างไร:** + -- ไฟล์ฮาร์ทบีทรันไทม์พร้อม PID, การประทับเวลา, การขนส่ง, จำนวนเครื่องมือ และโหมดขอบเขต -- API สถานะ MCP รวมการเต้นของหัวใจ + กิจกรรมล่าสุด -- การ์ดสถานะ UI สำหรับความสดใหม่ของกระบวนการ/สถานะการออนไลน์/การเต้นของหัวใจ +
+🤖 16. "I want to control model behavior globally" -<รายละเอียด> -📋 20. "ฉันต้องการใช้เครื่องมือ MCP ที่ตรวจสอบได้" +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -เมื่อเครื่องมือเปลี่ยนแปลงการกำหนดค่าหรือทริกเกอร์การดำเนินการ ทีมจำเป็นต้องมีการตรวจสอบย้อนกลับทางนิติเวช +**How OmniRoute solves it:** -**OmniRoute แก้ปัญหาอย่างไร:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- การบันทึกการตรวจสอบที่ได้รับการสนับสนุนจาก SQLite สำหรับการเรียกใช้เครื่องมือ MCP -- กรองตามเครื่องมือ ความสำเร็จ/ล้มเหลว คีย์ API และการแบ่งหน้า -- ตารางการตรวจสอบแดชบอร์ด + จุดสิ้นสุดสถิติสำหรับระบบอัตโนมัติ
+ -<รายละเอียด> -🔐 21. "ฉันต้องการสิทธิ์ MCP ที่มีขอบเขตต่อการบูรณาการ" +
+🧰 17. "I need MCP tools as first-class product capabilities" -ไคลเอนต์ที่แตกต่างกันควรมีสิทธิ์เข้าถึงหมวดหมู่เครื่องมือน้อยที่สุด +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**OmniRoute แก้ปัญหาอย่างไร:** +**How OmniRoute solves it:** -- ขอบเขต MCP แบบละเอียด 10 รายการสำหรับการเข้าถึงเครื่องมือที่ควบคุม -- การบังคับใช้ขอบเขตและการมองเห็นใน UI การจัดการ MCP -- ท่าทางเริ่มต้นที่ปลอดภัยสำหรับเครื่องมือในการปฏิบัติงาน
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding -<รายละเอียด> -⚙️ 22. "ฉันต้องการการควบคุมการปฏิบัติงานโดยไม่ต้องปรับใช้ใหม่" + -ทีมต้องการการเปลี่ยนแปลงรันไทม์อย่างรวดเร็วระหว่างเหตุการณ์หรือเหตุการณ์ต้นทุน +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**OmniRoute แก้ปัญหาอย่างไร:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- สลับการเปิดใช้งานคอมโบโดยตรงจากแดชบอร์ด MCP -- ใช้โปรไฟล์ความยืดหยุ่นจากชุดนโยบายที่กำหนดไว้ล่วงหน้า -- รีเซ็ตสถานะเซอร์กิตเบรกเกอร์จากแผงการทำงานเดียวกัน
+**How OmniRoute solves it:** -<รายละเอียด> -🔄 23. "ฉันต้องการการมองเห็นและการยกเลิกวงจรการใช้งาน A2A แบบสด" +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -หากไม่มีการมองเห็นวงจรการใช้งาน เหตุการณ์ของงานจะยากต่อการคัดแยก + -**OmniRoute แก้ปัญหาอย่างไร:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- รายการงาน/การกรองตามสถานะ/ทักษะพร้อมการแบ่งหน้า -- เจาะลึกข้อมูลเมตาของงาน เหตุการณ์ และสิ่งประดิษฐ์ -- จุดสิ้นสุดการยกเลิกงานและการดำเนินการ UI พร้อมการยืนยัน
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. -<รายละเอียด> -🌊 24. "ฉันต้องการเมตริกสตรีมที่ใช้งานอยู่สำหรับโหลด A2A" +**How OmniRoute solves it:** -เวิร์กโฟลว์การสตรีมจำเป็นต้องมีข้อมูลเชิงลึกในการดำเนินงานเกี่ยวกับการทำงานพร้อมกันและการเชื่อมต่อแบบสด +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**OmniRoute แก้ปัญหาอย่างไร:** + -- ตัวนับสตรีมที่ใช้งานรวมอยู่ในสถานะ A2A -- การประทับเวลางานล่าสุดและการนับต่อรัฐ -- การ์ดแดชบอร์ด A2A สำหรับการตรวจสอบการปฏิบัติงานแบบเรียลไทม์ +
+📋 20. "I need auditable MCP tool execution" -<รายละเอียด> -🪪 25. "ฉันต้องการการค้นพบตัวแทนมาตรฐานสำหรับลูกค้า" +When tools mutate config or trigger ops actions, teams need forensic traceability. -ไคลเอนต์และผู้ควบคุมภายนอกต้องการเมตาดาต้าที่เครื่องอ่านได้เพื่อการเริ่มต้นใช้งาน +**How OmniRoute solves it:** -**OmniRoute แก้ปัญหาอย่างไร:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- บัตรตัวแทนเปิดเผยที่ `/.well-known/agent.json` -- ความสามารถและทักษะที่แสดงใน UI การจัดการ -- API สถานะ A2A รวมถึงข้อมูลเมตาการค้นพบสำหรับระบบอัตโนมัติ
+ -<รายละเอียด> -🧭 26. "ฉันต้องการการค้นพบโปรโตคอลใน UX ของผลิตภัณฑ์" +
+🔐 21. "I need scoped MCP permissions per integration" -หากผู้ใช้ไม่พบพื้นผิวของโปรโตคอล การนำไปใช้และคุณภาพการสนับสนุนจะลดลง +Different clients should have least-privilege access to tool categories. -**OmniRoute แก้ปัญหาอย่างไร:** +**How OmniRoute solves it:** -- หน้า**ปลายทาง**รวมพร้อมแท็บสำหรับตำแหน่งข้อมูลพร็อกซี, MCP, A2A และ API -- สลับสถานะบริการอินไลน์ (ออนไลน์/ออฟไลน์) สำหรับ MCP และ A2A -- ลิงก์จากภาพรวมไปยังแท็บการจัดการเฉพาะ
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling -<รายละเอียด> -🧪 27. "ฉันต้องการการตรวจสอบโปรโตคอลแบบ end-to-end กับไคลเอนต์จริง" + -การทดสอบจำลองไม่เพียงพอที่จะตรวจสอบความเข้ากันได้ของโปรโตคอลก่อนเผยแพร่ +
+⚙️ 22. "I need operational controls without redeploying" -**OmniRoute แก้ปัญหาอย่างไร:** +Teams need quick runtime changes during incidents or cost events. -- ชุด E2E ที่บูทแอปและใช้การขนส่งไคลเอนต์ MCP SDK จริง -- ไคลเอนต์ A2A ทดสอบการค้นหา ส่ง สตรีม รับ และยกเลิกโฟลว์ -- ยืนยันการตรวจสอบข้ามกับการตรวจสอบ MCP และ API งาน A2A
+**How OmniRoute solves it:** -<รายละเอียด> -📡 28. "ฉันต้องการความสามารถในการสังเกตแบบรวมศูนย์ในทุกอินเทอร์เฟซ" +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -การแยกความสามารถในการสังเกตตามโปรโตคอลทำให้เกิดจุดบอดและ MTTR ที่ยาวขึ้น + -**OmniRoute แก้ปัญหาอย่างไร:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- แดชบอร์ด/บันทึก/การวิเคราะห์แบบรวมในผลิตภัณฑ์เดียว -- สุขภาพ + การตรวจสอบ + ขอการตรวจวัดทางไกลผ่านเลเยอร์ OpenAI, MCP และ A2A -- API การดำเนินงานสำหรับสถานะและระบบอัตโนมัติ
+Without lifecycle visibility, task incidents become hard to triage. -<รายละเอียด> -💼 29. "ฉันต้องการรันไทม์หนึ่งรายการสำหรับพร็อกซี + เครื่องมือ + การจัดการเอเจนต์" +**How OmniRoute solves it:** -การใช้บริการแยกกันจำนวนมากจะเพิ่มต้นทุนการดำเนินงานและโหมดความล้มเหลว +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**OmniRoute แก้ปัญหาอย่างไร:** + -- พร็อกซีที่เข้ากันได้กับ OpenAI, เซิร์ฟเวอร์ MCP และเซิร์ฟเวอร์ A2A ในสแต็กเดียว -- การรับรองความถูกต้องที่ใช้ร่วมกัน ความยืดหยุ่น การจัดเก็บข้อมูล และความสามารถในการสังเกต -- รูปแบบนโยบายที่สอดคล้องกันในทุกรูปแบบการโต้ตอบ +
+🌊 24. "I need active stream metrics for A2A load" -<รายละเอียด> -🚀 30. "ฉันต้องจัดส่งเวิร์กโฟลว์แบบตัวแทนโดยไม่มีการแผ่ขยายโค้ดกาว" +Streaming workflows require operational insight into concurrency and live connections. -ทีมจะสูญเสียความเร็วเมื่อรวมบริการและสคริปต์เฉพาะกิจหลายรายการเข้าด้วยกัน +**How OmniRoute solves it:** -**OmniRoute แก้ปัญหาอย่างไร:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- กลยุทธ์อุปกรณ์ปลายทางแบบครบวงจรสำหรับลูกค้าและตัวแทน -- UIs การจัดการโปรโตคอลในตัวและเส้นทางการตรวจสอบควัน -- รากฐานที่พร้อมสำหรับการผลิต (ความปลอดภัย การบันทึก ความยืดหยุ่น การสำรองข้อมูล)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Playbook A: เพิ่มการสมัครสมาชิกแบบชำระเงินให้สูงสุด + การสำรองข้อมูลราคาถูก**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -608,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Playbook B: สแต็กการเข้ารหัสที่ไม่มีค่าใช้จ่าย**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Playbook C: ห่วงโซ่ทางเลือกที่เปิดตลอด 24 ชั่วโมงทุกวัน**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -631,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Playbook D: ตัวแทนดำเนินการด้วย MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> ตั้งค่าการเข้ารหัส AI ในไม่กี่นาทีที่**$0/เดือน**เชื่อมต่อบัญชีฟรีเหล่านี้และใช้คอมโบ**Free Stack**ในตัว +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| ขั้นตอน | การกระทำ | ผู้ให้บริการปลดล็อคแล้ว | +| Step | Action | Providers Unlocked | | ---- | -------------------------------------------------- | ------------------------------------------------------------------ | -| 1 | เชื่อมต่อ**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**ไม่จำกัด**| -| 2 | เชื่อมต่อ**Qoder**(Google OAuth) | kimi-k2-คิด, qwen3-coder-plus, deepseek-r1... —**ไม่จำกัด**| -| 3 | เชื่อมต่อ**Qwen**(รหัสอุปกรณ์) | qwen3-coder-plus, qwen3-coder-flash... —**ไม่จำกัด**| -| 4 | เชื่อมต่อ**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**ฟรี 180K/เดือน**| -| 5 | `/dashboard/combos` →**เทมเพลตสแต็คฟรี ($0)**| Round-robin ผู้ให้บริการฟรีทั้งหมดโดยอัตโนมัติ | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**ชี้ IDE/CLI ใดๆ ไปที่:**`http://localhost:20128/v1` · คีย์ API: `any-string` · เสร็จสิ้น +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**ความคุ้มครองเพิ่มเติมเพิ่มเติม (ฟรี):**คีย์ Groq API (ฟรี 30 RPM), NVIDIA NIM (ฟรี 40 RPM, รุ่น 70+), Cerebras (1M tok/วัน), คีย์ LongCat API (โทเค็น 50M/วัน!), Cloudflare Workers AI (10K Neurons/วัน, รุ่น 50+)## เริ่มต้นอย่างรวดเร็ว +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## เริ่มต้นอย่างรวดเร็ว ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **ผู้ใช้ pnpm:**เรียกใช้ `pnpm Approve-builds -g` หลังจากติดตั้งเพื่อเปิดใช้งานสคริปต์บิลด์เนทิฟที่ `better-sqlite3` และ `@swc/core` ต้องการ: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > -> ```ทุบตี -> pnpm ติดตั้ง -g omniroute -> pnpm อนุมัติสร้าง -g # เลือกแพ็คเกจทั้งหมด → อนุมัติ -> ทุกเส้นทาง +> ```bash +> pnpm install -g omniroute +> pnpm approve-builds -g # Select all packages → approve +> omniroute > ``` -แดชบอร์ดเปิดที่ `http://localhost:20128` และ URL ฐาน API คือ `http://localhost:20128/v1` +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| คำสั่ง | คำอธิบาย | -| -------------------------- | --------------------------------------------------------------- | -| `ทุกเส้นทาง` | เริ่มเซิร์ฟเวอร์ (`PORT=20128`, API และแดชบอร์ดบนพอร์ตเดียวกัน) | -| `ทุกเส้นทาง -- พอร์ต 3000` | ตั้งค่าพอร์ต Canonical/API เป็น 3000 | -| `ทุกเส้นทาง --mcp` | เริ่มเซิร์ฟเวอร์ MCP (การขนส่ง stdio) | -| `ทุกเส้นทาง --ไม่เปิด` | อย่าเปิดเบราว์เซอร์อัตโนมัติ | -| `ทุกเส้นทาง --help` | แสดงความช่วยเหลือ | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -โหมดแยกพอร์ตเสริม:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -สำหรับการปรับใช้ส่วนใหญ่ คุณเพียงต้องการ: +For most deployments, you only need: -| ตัวแปร | ค่าเริ่มต้น | วัตถุประสงค์ | -| ------------------------ | --------------------------------- | ------------------------------------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600000` | เส้นพื้นฐานที่ใช้ร่วมกันสำหรับการดึงข้อมูลอัปสตรีม การหมดเวลา Undici ที่ซ่อนอยู่ คำขอลายนิ้วมือ TLS และคำขอบริดจ์ API/การหมดเวลาพร็อกซี | -| `STREAM_IDLE_TIMEOUT_MS` | สืบทอด `REQUEST_TIMEOUT_MS` | ช่องว่างสูงสุดระหว่างการสตรีมชิ้นส่วนก่อนที่ OmniRoute จะยกเลิกสตรีม SSE | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -ความเข้ากันได้แบบย้อนหลังยังคงอยู่: `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` ที่มีอยู่ และ vars การหมดเวลาต่อเลเยอร์อื่นๆ ที่มีอยู่ยังคงใช้งานได้และลบล้างพื้นฐานที่ใช้ร่วมกัน +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -การแทนที่ขั้นสูงจะพร้อมใช้งานหากคุณต้องการการควบคุมที่ละเอียดยิ่งขึ้น:| ตัวแปร | ค่าเริ่มต้น | วัตถุประสงค์ | -| -------------------------------------------- | ----------------------------------------------- | -------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | สืบทอด `REQUEST_TIMEOUT_MS` | การหมดเวลาคำขออัปสตรีมทั้งหมดที่ใช้โดยสัญญาณยกเลิกการดึงข้อมูลหลัก | -| `FETCH_HEADERS_TIMEOUT_MS` | สืบทอด `FETCH_TIMEOUT_MS` | Undici กำหนดเวลาในการรับส่วนหัวการตอบกลับต้นทาง | -| `FETCH_BODY_TIMEOUT_MS` | สืบทอด `FETCH_TIMEOUT_MS` | Undici กำหนดเวลาระหว่างชิ้นส่วนเนื้อหาต้นน้ำ (`0` ปิดใช้งาน) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP หมดเวลาการเชื่อมต่อ | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici การหมดเวลาซ็อกเก็ต Keep-alive ที่ไม่ได้ใช้งาน | -| `TLS_CLIENT_TIMEOUT_MS` | สืบทอด `FETCH_TIMEOUT_MS` | หมดเวลาสำหรับคำขอลายนิ้วมือ TLS ที่ดำเนินการผ่าน `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | สืบทอด `REQUEST_TIMEOUT_MS` หรือ `30000` | หมดเวลาสำหรับการส่งต่อพร็อกซี `/ v1` จากพอร์ต API ไปยังพอร์ตแดชบอร์ด | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `สูงสุด(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | การหมดเวลาคำขอขาเข้าบนเซิร์ฟเวอร์บริดจ์ API | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | การหมดเวลาส่วนหัวขาเข้าบนเซิร์ฟเวอร์บริดจ์ API | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | การหมดเวลา Keep-alive บนเซิร์ฟเวอร์บริดจ์ API | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | การหมดเวลาการไม่ใช้งานซ็อกเก็ตบนเซิร์ฟเวอร์บริดจ์ API (`0` ปิดการใช้งาน) | +Advanced overrides are available if you need finer control: -หากคุณเรียกใช้ OmniRoute หลัง Nginx, Caddy, Cloudflare หรือ Reverse Proxy อื่น ตรวจสอบให้แน่ใจว่าพร็อกซีนั้น -การหมดเวลายังสูงกว่าการหมดเวลาการสตรีม/การดึงข้อมูล OmniRoute ของคุณอีกด้วย### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. เปิดแดชบอร์ด → `ผู้ให้บริการ` และเชื่อมต่อผู้ให้บริการอย่างน้อยหนึ่งราย (คีย์ OAuth หรือ API) -2. เปิดแดชบอร์ด → `ปลายทาง` และสร้างคีย์ API -3. (ทางเลือก) เปิดแดชบอร์ด → `คอมโบ` และตั้งค่าห่วงโซ่ทางเลือกของคุณ### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -ทำงานร่วมกับ Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode และ SDK ที่เข้ากันได้กับ OpenAI### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (สำหรับการดำเนินการที่ขับเคลื่อนด้วยเครื่องมือ):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -จากนั้นเชื่อมต่อไคลเอนต์ MCP ของคุณผ่าน `stdio` และเครื่องมือทดสอบเช่น: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (สำหรับเวิร์กโฟลว์ตัวแทนถึงตัวแทน):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -760,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -ชุดนี้ตรวจสอบการไหลของไคลเอ็นต์ MCP และ A2A จริงกับแอปที่ทำงานอยู่### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -768,14 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` -<รายละเอียด> +
+Void Linux (`xbps-src` template) -Void Linux (เทมเพลต `xbps-src`) - -สำหรับผู้ใช้ Void Linux คุณสามารถสร้างแพ็คเกจดั้งเดิมได้โดยใช้ `xbps-src` บันทึกบล็อกนี้เป็น `srcpkgs/omniroute/template`:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -787,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -795,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -871,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -882,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute มีให้บริการในรูปแบบอิมเมจ Docker สาธารณะบน [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute) +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**วิ่งด่วน:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -892,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**พร้อมไฟล์สภาพแวดล้อม:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**การใช้นักเทียบท่าเขียน:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -การสนับสนุนแดชบอร์ดสำหรับการปรับใช้ Docker ขณะนี้ได้รวม**Cloudflare Quick Tunnel**ในคลิกเดียวบน `Dashboard → Endpoints` ครั้งแรกที่เปิดใช้งานการดาวน์โหลด `cloudflared` เมื่อจำเป็นเท่านั้น เริ่มทันเนลชั่วคราวไปยังตำแหน่งข้อมูล `/v1` ปัจจุบันของคุณ และแสดง URL `https://*.trycloudflare.com/v1` ที่สร้างขึ้นด้านล่าง URL สาธารณะปกติของคุณโดยตรง +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -หมายเหตุ: +Notes: -- URL ของอุโมงค์ด่วนเป็นแบบชั่วคราวและเปลี่ยนแปลงหลังจากการรีสตาร์ททุกครั้ง -- Quick Tunnels จะไม่ถูกกู้คืนอัตโนมัติหลังจากการรีสตาร์ท OmniRoute หรือคอนเทนเนอร์ เปิดใช้งานอีกครั้งจากแดชบอร์ดเมื่อจำเป็น -- ปัจจุบันการติดตั้งแบบมีการจัดการรองรับ Linux, macOS และ Windows บน `x64` / `arm64` -- Quick Tunnels ที่มีการจัดการมีค่าเริ่มต้นเป็นการขนส่ง HTTP/2 เพื่อหลีกเลี่ยงคำเตือนบัฟเฟอร์ QUIC UDP ที่มีเสียงดังในสภาพแวดล้อมคอนเทนเนอร์ที่จำกัด ตั้งค่า `CLOUDFLARED_PROTOCOL=quic` หรือ `auto` หากคุณต้องการการขนส่งอื่น -- อิมเมจ Docker รวมรูท CA ของระบบแล้วส่งต่อไปยัง `cloudflared` ที่ได้รับการจัดการ ซึ่งหลีกเลี่ยงความล้มเหลวในการเชื่อถือ TLS เมื่อทันเนลบูตสแตรปภายในคอนเทนเนอร์ -- SQLite ทำงานในโหมด WAL `นักเทียบท่าหยุด` ควรได้รับอนุญาตให้เสร็จสิ้นเพื่อให้ OmniRoute สามารถตรวจสอบการเปลี่ยนแปลงล่าสุดกลับเข้าไปใน `storage.sqlite` -- ไฟล์เขียนที่รวมมาได้ตั้งค่าระยะเวลาผ่อนผันการหยุด 40 วินาทีแล้ว หากคุณเรียกใช้อิมเมจโดยตรง ให้คง `--stop-timeout 40` (หรือที่คล้ายกัน) ไว้ เพื่อที่การหยุดแบบแมนนวลจะไม่ตัดการล้างการปิดระบบ -- ตั้งค่า `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` หากคุณต้องการให้ OmniRoute ใช้ไบนารี่ที่มีอยู่แทนการดาวน์โหลด +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**การใช้ Docker Compose กับแคดดี้ (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute สามารถเปิดเผยได้อย่างปลอดภัยโดยใช้การจัดเตรียม SSL อัตโนมัติของ Caddy ตรวจสอบให้แน่ใจว่าระเบียน DNS A ของโดเมนของคุณชี้ไปที่ IP ของเซิร์ฟเวอร์ของคุณ```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` - -| รูปภาพ | แท็ก | ขนาด | คำอธิบาย | +| Image | Tag | Size | Description | | ------------------------ | -------- | ------ | --------------------- | -| `diegosouzapw/ทุกเส้นทาง` | `ล่าสุด` | ~250MB | รุ่นเสถียรล่าสุด | -| `diegosouzapw/ทุกเส้นทาง` | `1.0.3` | ~250MB | เวอร์ชันปัจจุบัน |--- +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | + +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**ใหม่!**OmniRoute พร้อมใช้งานแล้วในรูปแบบ**แอปพลิเคชันเดสก์ท็อปดั้งเดิม**สำหรับ Windows, macOS และ Linux +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -เรียกใช้ OmniRoute เป็นแอปเดสก์ท็อปแบบสแตนด์อโลน ไม่ต้องใช้เทอร์มินัล ไม่ต้องใช้เบราว์เซอร์ ไม่ต้องใช้อินเทอร์เน็ตสำหรับรุ่นท้องถิ่น แอพที่ใช้อิเล็กตรอนประกอบด้วย: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**หน้าต่างดั้งเดิม**— หน้าต่างแอปเฉพาะพร้อมการรวมถาดระบบ -- 🔄**เริ่มอัตโนมัติ**— เปิด OmniRoute เมื่อเข้าสู่ระบบ -- 🔔**การแจ้งเตือนแบบเนทีฟ**— รับการแจ้งเตือนเกี่ยวกับโควต้าหมดหรือปัญหาของผู้ให้บริการ -- ⚡**ติดตั้งเพียงคลิกเดียว**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**โหมดออฟไลน์**— ทำงานแบบออฟไลน์โดยสมบูรณ์กับเซิร์ฟเวอร์รวม### เริ่มต้นอย่างรวดเร็ว +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### เริ่มต้นอย่างรวดเร็ว ```bash # Development mode @@ -981,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -เมื่อย่อเล็กสุด OmniRoute จะอยู่ในถาดระบบของคุณด้วยการดำเนินการด่วน: +When minimized, OmniRoute lives in your system tray with quick actions: -- เปิดแดชบอร์ด -- เปลี่ยนพอร์ตเซิร์ฟเวอร์ -- ออกจากแอปพลิเคชัน +- Open dashboard +- Change server port +- Quit application -📖 เอกสารฉบับเต็ม: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| ชั้น | ผู้ให้บริการ | ราคา | รีเซ็ตโควต้า | ดีที่สุดสำหรับ | -| ------------------ | --------------------------------- | ------------------------------ | ------------------------------- | --------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 สมัครสมาชิก** | รหัสคลอดด์ (Pro) | $20/เดือน | 5 ชม. + รายสัปดาห์ | สมัครสมาชิกแล้ว | -| | Codex (พลัส/โปร) | $20-200/เดือน | 5 ชม. + รายสัปดาห์ | ผู้ใช้ OpenAI | -| | ราศีเมถุน CLI | **ฟรี** | 180K/เดือน + 1K/วัน | ทุกคน! | -| | นักบิน GitHub | $10-19/เดือน | รายเดือน | ผู้ใช้ GitHub | -| **🔑 คีย์ API** | NVIDIA NIM | **ฟรี**(พัฒนาตลอดไป) | ~40 รอบต่อนาที | รุ่นเปิดมากกว่า 70 รุ่น | -| | สมอง | **ฟรี**(1M ต๊อก/วัน) | 60,000 ทีพีเอ็ม / 30 รอบต่อนาที | เร็วที่สุดในโลก | -| | กรอค | **ฟรี**(30 รอบต่อนาที) | 14.4K RPD | ลามะ/เจมม่าที่เร็วเป็นพิเศษ | -| | DeepSeek V3.2 | $0.27/$1.10 ต่อ 1M | ไม่มี | เหตุผลด้านราคา/คุณภาพที่ดีที่สุด | -| | xAI Grok-4 เร็ว | **$0.20/$0.50 ต่อ 1M**🆕 | ไม่มี | เร็วที่สุด + การเรียกเครื่องมือ, ต่ำมาก | -| | xAI Grok-4 (มาตรฐาน) | $0.20/$1.50 ต่อ 1M 🆕 | ไม่มี | การใช้เหตุผลเป็นเรือธงจาก xAI | -| | มิสทรัล | ทดลองใช้ฟรี + จ่ายเงิน | อัตราจำกัด | AI ยุโรป | -| | OpenRouter | จ่ายตามการใช้งาน | ไม่มี | มีโมเดลมากกว่า 100 แบบ | -| **💰 ราคาถูก** | GLM-5 (ผ่าน Z.AI) 🆕 | $0.5/1M | ทุกวัน 10.00 น. | เอาต์พุต 128K เรือธงใหม่ล่าสุด | -| | GLM-4.7 | $0.6/1M | ทุกวัน 10.00 น. | สำรองงบประมาณ | -| | MiniMax M2.5 🆕 | $0.3/1M อินพุต | กลิ้ง 5 ชั่วโมง | การใช้เหตุผล + งานตัวแทน | -| | MiniMax M2.1 | $0.2/1M | กลิ้ง 5 ชั่วโมง | ตัวเลือกที่ถูกที่สุด | -| | Kimi K2.5 (Moonshot API) 🆕 | จ่ายตามการใช้งาน | ไม่มี | การเข้าถึง Moonshot API โดยตรง | -| | คิมิ K2 | $9/เดือน คงที่ | 10M โทเค็น/เดือน | ต้นทุนที่คาดการณ์ได้ | -| **🆓 ฟรี** | คิวเดอร์ | **$0** | ไม่จำกัด | 5 รุ่นไม่จำกัด | -| | ควีน | **$0** | ไม่จำกัด | 4 รุ่นไม่จำกัด | -| | คิโระ | **$0** | ไม่จำกัด | Claude Sonnet/ไฮกุ (AWS Builder) | -| | LongCat Flash-Lite 🆕 | **$0**(50M ต็อก/วัน 🔥) | 1 RPS | โควต้าฟรีที่ใหญ่ที่สุดในโลก | -| | AI การผสมเกสร 🆕 | **$0**(ไม่ต้องใช้กุญแจ) | 1 คำขอ/15 วินาที | GPT-5, Claude, DeepSeek, ลามะ 4 | -| | AI ของผู้ปฏิบัติงาน Cloudflare 🆕 | **$0**(10,000 เซลล์ประสาท/วัน) | ~150 ครั้ง/วัน | โมเดลมากกว่า 50 แบบ ขอบระดับโลก | -| | สเกลเวย์ AI 🆕 | **$0**(รวมโทเค็น 1M) | อัตราจำกัด | EU/GDPR, Qwen3 235B, ลามะ 70B | > 🆕**เพิ่มโมเดลใหม่ (มี.ค. 2026):**Grok-4 Fast family ที่ $0.20/$0.50/M (เปรียบเทียบที่ 1143ms — เร็วกว่า Gemini 2.5 Flash 30%), GLM-5 ผ่าน Z.AI พร้อมเอาต์พุต 128K, การใช้เหตุผล MiniMax M2.5, ราคาที่อัปเดต DeepSeek V3.2, Kimi K2.5 ผ่าน Moonshot direct API | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 $0 Combo Stack — การตั้งค่าที่สมบูรณ์ฟรี:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**ไม่มีค่าใช้จ่าย ไม่หยุดเขียนโค้ด**กำหนดค่านี้เป็น OmniRoute เดียวและทางเลือกทั้งหมดจะเกิดขึ้นโดยอัตโนมัติ ไม่มีการสลับด้วยตนเองเลย--- +--- --- ## 🆓 Free Models — What You Actually Get -> ทุกรุ่นด้านล่าง**ฟรี 100% โดยไม่ต้องใช้บัตรเครดิต**OmniRoute จะกำหนดเส้นทางอัตโนมัติระหว่างกันเมื่อโควต้าหนึ่งหมด — รวมเข้าด้วยกันเพื่อคอมโบ $0 ที่ไม่อาจแตกหักได้### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| รุ่น | คำนำหน้า | ขีดจำกัด | ขีดจำกัดอัตรา | +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | --------------------- | -| `โคลด-โคลง-4.5` | `kr/` |**ไม่จำกัด**| ไม่มีรายงานสูงสุดรายวัน | -| `claude-ไฮกุ-4.5` | `kr/` |**ไม่จำกัด**| ไม่มีรายงานสูงสุดรายวัน | -| `claude-opus-4.6` | `kr/` |**ไม่จำกัด**| บทประพันธ์ล่าสุดผ่าน Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -| รุ่น | คำนำหน้า | ขีดจำกัด | ขีดจำกัดอัตรา | +### 🟢 QODER MODELS (Free PAT via qodercli) + +| Model | Prefix | Limit | Rate Limit | | ------------------ | ------ | ------------- | --------------- | -| `kimi-k2-คิด` | `ถ้า/` |**ไม่จำกัด**| ไม่มีรายงานหมวก | -| `qwen3-coder-plus` | `ถ้า/` |**ไม่จำกัด**| ไม่มีรายงานหมวก | -| `ดีพซีค-r1` | `ถ้า/` |**ไม่จำกัด**| ไม่มีรายงานหมวก | -| `มินิแม็กซ์-m2.1` | `ถ้า/` |**ไม่จำกัด**| ไม่มีรายงานหมวก | -| `คิมิ-k2` | `ถ้า/` |**ไม่จำกัด**| ไม่มีรายงานหมวก | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -> วิธีการเชื่อมต่อที่แนะนำ:**โทเค็นการเข้าถึงส่วนบุคคล + `qodercli`**เบราว์เซอร์ OAuth คือ -> ทดลองและปิดใช้งานตามค่าเริ่มต้น เว้นแต่จะมีการกำหนดค่าตัวแปรสภาพแวดล้อม `QODER_OAUTH_*`### 🟡 QWEN MODELS (Device Code Auth) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| รุ่น | คำนำหน้า | ขีดจำกัด | ขีดจำกัดอัตรา | +### 🟡 QWEN MODELS (Device Code Auth) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | ------------------- | -| `qwen3-coder-plus` | `คิว/` |**ไม่จำกัด**| ไม่มีรายงานหมวก | -| `qwen3-coder-flash` | `คิว/` |**ไม่จำกัด**| ไม่มีรายงานหมวก | -| `qwen3-coder-ถัดไป` | `คิว/` |**ไม่จำกัด**| ไม่มีรายงานหมวก | -| `แบบจำลองวิสัยทัศน์` | `คิว/` |**ไม่จำกัด**| ต่อเนื่องหลายรูปแบบ (ภาพ) |### 🟣 GEMINI CLI (Google OAuth) +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| รุ่น | คำนำหน้า | ขีดจำกัด | ขีดจำกัดอัตรา | -| ------------------------ | ------ | ------------------------------- | ------------- | -| `gemini-3-flash-preview` | `gc/` |**180K tok/เดือน**+ 1K/วัน | รีเซ็ตรายเดือน | -| `เจมินี่-2.5-โปร` | `gc/` | 180K/เดือน (พูลรวม) | คุณภาพสูง |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +### 🟣 GEMINI CLI (Google OAuth) -| ชั้น | ขีดจำกัดรายวัน | ขีดจำกัดอัตรา | หมายเหตุ | +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | + +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---------- | ------------ | ----------- | ------------------------------------------------------ | -| ฟรี (Dev) | ไม่มีฝาปิดโทเค็น |**~40 รอบต่อนาที**| มากกว่า 70 รุ่น; การเปลี่ยนไปใช้ขีดจำกัดอัตราเดียวในช่วงกลางปี ​​2025 | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -รุ่นฟรียอดนิยม: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -| ชั้น | ขีดจำกัดรายวัน | ขีดจำกัดอัตรา | หมายเหตุ | -| ---- | ----------------- | ---------------- | ----------------------------------------------- | -| ฟรี |**1M โทเค็น/วัน**| 60,000 ทีพีเอ็ม / 30 รอบต่อนาที | การอนุมาน LLM ที่เร็วที่สุดในโลก รีเซ็ตทุกวัน | +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -มีให้บริการฟรี: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -| ชั้น | ขีดจำกัดรายวัน | ขีดจำกัดอัตรา | หมายเหตุ | -| ---- | ------------- | ---------------- | --------------------------------------------- | -| ฟรี |**14.4K RPD**| 30 RPM ต่อรุ่น | ไม่มีบัตรเครดิต 429 ตามวงเงิน ไม่คิดเงิน | +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -มีให้บริการฟรี: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +### 🔴 GROQ (Free API Key — console.groq.com) -| รุ่น | คำนำหน้า | โควต้าฟรีรายวัน | หมายเหตุ | -| --------------------------------- | ------ | ----------------- | ----------------------- | -| `LongCat-Flash-Lite` | `lc/` |**โทเค็น 50M**💥 | โควต้าฟรีที่ใหญ่ที่สุดเท่าที่เคยมีมา | -| `LongCat-Flash-Chat` | `lc/` | โทเค็น 500K | แชทหลายรอบ | -| `LongCat-Flash-คิด` | `lc/` | โทเค็น 500K | การใช้เหตุผล / CoT | -| `LongCat-Flash-Thinking-2601` | `lc/` | โทเค็น 500K | เวอร์ชันม.ค. 2569 | -| `LongCat-Flash-Omni-2603` | `lc/` | โทเค็น 500K | ต่อเนื่องหลายรูปแบบ | +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ------------- | ---------------- | ----------------------------------------- | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -> ฟรี 100% ขณะอยู่ในช่วงเบต้าสาธารณะ ลงทะเบียนได้ที่ [longcat.chat](https://longcat.chat) ด้วยอีเมลหรือโทรศัพท์ รีเซ็ตทุกวัน 00:00 UTC### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -| รุ่น | คำนำหน้า | ขีดจำกัดอัตรา | ผู้ให้บริการเบื้องหลัง | +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 + +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | + +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. + +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| `เปิดใจ` | `พล/` | 1 คำขอ/15 วินาที | GPT-5 | -| `โคลด` | `พล/` | 1 คำขอ/15 วินาที | มานุษยวิทยาคลอด | -| `ราศีเมถุน` | `พล/` | 1 คำขอ/15 วินาที | Google ราศีเมถุน | -| `แสวงหาอย่างลึกซึ้ง' | `พล/` | 1 คำขอ/15 วินาที | DeepSeek V3 | -| `ลามะ` | `พล/` | 1 คำขอ/15 วินาที | Meta Llama 4 ลูกเสือ | -| `มิสทรัล` | `พล/` | 1 คำขอ/15 วินาที | มิสทรัล AI | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**แรงเสียดทานเป็นศูนย์:**ไม่ต้องสมัคร ไม่มีคีย์ API เพิ่มผู้ให้บริการ Pollinations ด้วยฟิลด์คีย์ว่าง และใช้งานได้ทันที### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| ชั้น | เซลล์ประสาทรายวัน | การใช้งานที่เทียบเท่า | หมายเหตุ | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 + +| Tier | Daily Neurons | Equivalent Usage | Notes | | ---- | ------------- | --------------------------------------- | ----------------------- | -| ฟรี |**10,000**| ~150 LLM resp / เสียง 500s / การฝัง 15K | ขอบโลก รุ่น 50+ | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -รุ่นฟรียอดนิยม: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (เสียงฟรี!), `@cf/qwen/qwen2.5-coder-15b-instruct` +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -> ต้องใช้โทเค็น API + รหัสบัญชีจาก [dash.cloudflare.com](https://dash.cloudflare.com) รหัสบัญชีร้านค้าในการตั้งค่าผู้ให้บริการ### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -| ชั้น | โควต้าฟรี | ที่ตั้ง | หมายเหตุ | +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | | ---- | ------------- | ------------ | ----------------------------------- | -| ฟรี |**โทเค็น 1 ล้าน**| 🇫🇷 ปารีส สหภาพยุโรป | ไม่ต้องใช้บัตรเครดิตภายในวงเงิน | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | -ใช้ได้ฟรี: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` -> สอดคล้องกับสหภาพยุโรป/GDPR รับคีย์ API ได้ที่ [console.scaleway.com](https://console.scaleway.com) +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). ->**💡 สุดยอดสแต็คฟรี (ผู้ให้บริการ 11 ราย, $0 ตลอดไป):** +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> คิโระ (kr/) → คล็อด ซอนเน็ต/ไฮกุ ไม่จำกัด -> Qoder (ถ้า/) → kimi-k2-คิด, qwen3-coder-plus, deepseek-r1 ไม่จำกัด -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M โทเค็น/วัน 🔥 -> การผสมเกสร (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — ไม่ต้องใช้คีย์ -> Qwen (qw/) → รุ่น qwen3-coder ไม่จำกัด -> Gemini (ราศีเมถุน/) → Gemini 2.5 Flash — ฟรี 1,500 req/วัน -> Cloudflare AI (cf/) → โมเดล 50+ — 10,000 เซลล์ประสาท/วัน -> Scaleway (scw/) → Qwen3 235B, Llama 70B — โทเค็นฟรี 1M (EU) -> Groq (groq/) → Llama/Gemma — 14.4K req/วัน รวดเร็วเป็นพิเศษ -> NVIDIA NIM (nvidia/) → รุ่นเปิดมากกว่า 70 รุ่น — 40 RPM ตลอดไป -> Cerebras (cerebras/) → Llama/Qwen เร็วที่สุดในโลก — 1M tok/วัน -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> ถอดเสียง/วิดีโอใดๆ ในราคา**$0**— Deepgram Leads พร้อมฟรี $200, AssemblyAI สำรอง $50, Groq Whisper เป็นการสำรองข้อมูลฉุกเฉินไม่จำกัด +## 🎙️ Free Transcription Combo -| ผู้ให้บริการ | เครดิตฟรี | โมเดลที่ดีที่สุด | ขีดจำกัดอัตรา | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. + +| Provider | Free Credits | Best Model | Rate Limit | | ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | -| 🟢**ดีปแกรม**|**ฟรี $200**(สมัครสมาชิก) | `nova-3` — แม่นยำที่สุด 30+ ภาษา | ไม่มีการจำกัด RPM สำหรับเครดิตฟรี | -| 🔵**AssemblyAI**|**ฟรี $50**(สมัครสมาชิก) | `universal-3-pro` — บท, ความรู้สึก, PII | ไม่มีการจำกัด RPM สำหรับเครดิตฟรี | -| 🔴**โกรก**|**ฟรีตลอดไป**| `กระซิบ-ขนาดใหญ่-v3` — OpenAI กระซิบ | 30 RPM (อัตราจำกัด) | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | -**คำสั่งผสมที่แนะนำใน `/dashboard/combos`:**``` +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -จากนั้นใน `/dashboard/media` → แท็บ**การถอดเสียง**: อัปโหลดไฟล์เสียงหรือวิดีโอ → เลือกจุดสิ้นสุดคอมโบของคุณ → รับการถอดเสียงในรูปแบบที่รองรับ## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 ถูกสร้างขึ้นเพื่อเป็นแพลตฟอร์มการดำเนินงาน ไม่ใช่แค่รีเลย์พร็อกซี### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| คุณสมบัติ | มันทำอะไร | -| ------------------------------------------------ | ------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Grok-4 ฟาสต์แฟมิลี่** | รุ่น xAI ที่ $0.20/$0.50/M — เปรียบเทียบ 1143ms (เร็วกว่า Gemini 2.5 Flash 30%) | -| 🧠**GLM-5 ผ่าน Z.AI** | บริบทเอาต์พุต 128K, $0.5/1M — รุ่นเรือธงใหม่ล่าสุดจากตระกูล GLM | -| 🔮**MiniMax M2.5** | การใช้เหตุผล + งานเอเจนต์ที่ 0.30 ดอลลาร์/1 ล้าน — อัปเกรดอย่างมีนัยสำคัญจาก M2.1 | -| 🎯**toolCalling Flag ต่อรุ่น** | `toolCalling: true/false` ต่อรุ่นในรีจิสตรี — AutoCombo ข้ามโมเดลที่ไม่ใช้เครื่องมือ | -| 🌍**การตรวจจับเจตนาหลายภาษา** | คีย์เวิร์ด PT/ZH/ES/AR ในการให้คะแนน AutoCombo — การเลือกโมเดลที่ดีกว่าสำหรับเนื้อหาที่ไม่ใช่ภาษาอังกฤษ | -| 📊**ทางเลือกสำรองที่ขับเคลื่อนด้วยเกณฑ์มาตรฐาน** | เวลาแฝง p95 จริงจากคำขอสดฟีดการให้คะแนนคอมโบ — AutoCombo เรียนรู้จากข้อมูลจริง | -| 🔁**ขอการขจัดข้อมูลซ้ำซ้อน** | หน้าต่าง dedup ตามแฮชเนื้อหา — ปลอดภัยสำหรับหลายเอเจนต์ ป้องกันการเรียกเก็บเงินซ้ำ | -| 🔌**กลยุทธ์เราเตอร์แบบเสียบได้** | อินเทอร์เฟซ `RouterStrategy` ที่ขยายได้ — เพิ่มตรรกะการกำหนดเส้นทางที่กำหนดเองเป็นปลั๊กอิน | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| คุณสมบัติ | มันทำอะไร | -| ---------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**โมเดลสนามเด็กเล่น** | หน้าแดชบอร์ดเพื่อทดสอบโมเดลใดๆ โดยตรง — ตัวเลือกผู้ให้บริการ/โมเดล/ปลายทาง, Monaco Editor, การสตรีม, ยกเลิก, การกำหนดเวลา | -| 🔏**การจับคู่ลายนิ้วมือ CLI** | การจัดลำดับส่วนหัว/เนื้อหาต่อผู้ให้บริการเพื่อให้ตรงกับลายเซ็น CLI ดั้งเดิม — สลับตามผู้ให้บริการในการตั้งค่า > ความปลอดภัย**IP พร็อกซีของคุณยังคงอยู่** | -| 🤝**รองรับ ACP (โปรโตคอลไคลเอ็นต์ตัวแทน)** | การค้นพบเอเจนต์ CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw + อีก 9 รายการ), กระบวนการวางไข่, จุดสิ้นสุด `/api/acp/agents | -| 🤖**แดชบอร์ดตัวแทน ACP** | ดีบัก › หน้าตัวแทน — ตารางตัวแทน 14 รายพร้อมสถานะการติดตั้ง เวอร์ชัน แบบฟอร์มตัวแทนแบบกำหนดเองสำหรับเครื่องมือ CLI ผู้ใช้**OpenCode**จะได้รับปุ่ม "ดาวน์โหลด opencode.json" ที่สร้างการกำหนดค่าที่พร้อมใช้งานโดยอัตโนมัติสำหรับโมเดลที่มีอยู่ทั้งหมด | -| ????**การกำหนดเส้นทาง `apiFormat` โมเดลแบบกำหนดเอง** | โมเดลที่กำหนดเองที่มี `apiFormat: "responses"` กำหนดเส้นทางไปยังตัวแปล Responses API ได้อย่างถูกต้องแล้ว | -| 🏢**การแยกพื้นที่ทำงาน Codex** | พื้นที่ทำงาน Codex หลายรายการต่ออีเมล — OAuth แยกการเชื่อมต่ออย่างถูกต้องด้วยรหัสพื้นที่ทำงาน | -| 🔄**อัปเดตอิเล็กตรอนอัตโนมัติ** | แอปเดสก์ท็อปตรวจสอบการอัปเดต + ติดตั้งอัตโนมัติเมื่อรีสตาร์ท | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| คุณสมบัติ | มันทำอะไร | -| -------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🏽**เซิร์ฟเวอร์ MCP (25 เครื่องมือ)** | เครื่องมือ IDE/เอเจนต์ผ่าน 3 การขนส่ง: stdio, SSE (`/api/mcp/sse`), HTTP แบบสตรีมได้ (`/api/mcp/stream`) 18 คอร์ + 3 หน่วยความจำ + 4 เครื่องมือทักษะ | -| 🤝**เซิร์ฟเวอร์ A2A (JSON-RPC + SSE)** | การดำเนินการงานระหว่างเอเจนต์กับเอเจนต์ด้วยการซิงค์และการสตรีมโฟลว์ | -| 🧭**หน้าปลายทางรวม** | หน้าการจัดการแบบแท็บพร้อมแท็บ Endpoint Proxy, MCP, A2A และ API Endpoints | -| 🎚️**บริการเปิด/ปิดการสลับ** | สวิตช์เปิด/ปิดสำหรับ MCP และ A2A พร้อมการตั้งค่าคงอยู่ (ค่าเริ่มต้น: ปิด) | -| 🛰️**MCP รันไทม์ฮาร์ทบีท** | สถานะกระบวนการจริง (pid, สถานะการออนไลน์, อายุฮาร์ทบีท, การขนส่ง, โหมดขอบเขต) | -| 📋**เส้นทางการตรวจสอบ MCP** | บันทึกการตรวจสอบที่กรองได้พร้อมความสำเร็จ/ล้มเหลวและการระบุแหล่งที่มาที่สำคัญ | -| 🔐**การบังคับใช้ขอบเขต MCP** | สิทธิ์ขอบเขตแบบละเอียด 10 รายการสำหรับการเข้าถึงเครื่องมือที่ควบคุม | -| 📡**การจัดการวงจรงาน A2A** | แสดงรายการ/กรองงาน ตรวจสอบเหตุการณ์/สิ่งประดิษฐ์ ยกเลิกงานที่กำลังทำงานอยู่ | -| 📋**การค้นพบบัตรตัวแทน** | `/.well-known/agent.json` สำหรับการค้นพบอัตโนมัติของไคลเอ็นต์ | -| 🧪**ชุดทดสอบโปรโตคอล E2E** | ไคลเอนต์ MCP SDK + A2A จริงไหลใน `test:protocols:e2e` | -| ⚙️**การควบคุมการปฏิบัติงาน** | สลับคอมโบ ใช้โปรไฟล์ความยืดหยุ่น รีเซ็ตเบรกเกอร์จากพื้นผิวควบคุมเดียว | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| คุณสมบัติ | มันทำอะไร | -| ----------------------------------------- | --------------------------------------------------------------------------------- | ----------------------- | -| 🎯**ทางเลือกสำรองอัจฉริยะ 4 ระดับ** | เส้นทางอัตโนมัติ: การสมัครสมาชิก → คีย์ API → ถูก → ฟรี | -| 📊**การติดตามโควต้าแบบเรียลไทม์** | จำนวนโทเค็นสด + รีเซ็ตการนับถอยหลังต่อผู้ให้บริการ | -| 🔄**แปลรูปแบบ** | OpenAI ↔ Claude ↔ Gemini ↔ การตอบกลับด้วยการแปลงสคีมาที่ปลอดภัย | -| 👥**รองรับหลายบัญชี** | หลายบัญชีต่อผู้ให้บริการพร้อมตัวเลือกที่ชาญฉลาด | -| 🔄**รีเฟรชโทเค็นอัตโนมัติ** | โทเค็น OAuth รีเฟรชอัตโนมัติพร้อมลองอีกครั้ง | -| 🎨**คอมโบแบบกำหนดเอง** | 9 กลยุทธ์การปรับสมดุล + การควบคุมลูกโซ่สำรอง | -| 🌐**เราเตอร์ตัวแทน** | `ผู้ให้บริการ/*` การกำหนดเส้นทางแบบไดนามิก | -| 🧠**คิดควบคุมงบประมาณ** | ขีดจำกัดการให้เหตุผลแบบส่งผ่าน อัตโนมัติ แบบกำหนดเอง และแบบปรับเปลี่ยนได้ | -| 🔀**นามแฝงโมเดล** | นามแฝงโมเดลในตัว + แบบกำหนดเองและความปลอดภัยในการย้ายข้อมูล | -| ⚡**การเสื่อมสภาพของพื้นหลัง** | กำหนดเส้นทางงานพื้นหลังที่มีลำดับความสำคัญต่ำไปยังโมเดลที่ถูกกว่า | -| 🧪**การกำหนดเส้นทางอัจฉริยะที่รับรู้งาน** | เลือกโมเดลอัตโนมัติตามประเภทเนื้อหา (การเข้ารหัส/การมองเห็น/การวิเคราะห์/การสรุป) | -| 🔄**ขั้นตอนการทำงานของตัวแทน A2A** | ตัวกำหนด FSM ที่กำหนดสำหรับการเรียกใช้งานเอเจนต์แบบหลายขั้นตอนแบบมีสถานะ | -| 🔀**การกำหนดเส้นทางแบบปรับเปลี่ยนได้** | การแทนที่กลยุทธ์แบบไดนามิกตามปริมาณโทเค็นและความซับซ้อนของการแจ้งเตือน | -| 🎲**ความหลากหลายของผู้ให้บริการ** | การให้คะแนนเอนโทรปีของแชนนอนที่สมดุลกับการกระจายการรับส่งข้อมูลแบบคอมโบอัตโนมัติ | -| 💌**ระบบพร้อมฉีด** | มีการใช้การควบคุมพฤติกรรมทั่วโลกอย่างสม่ำเสมอ | -| 📄**ความเข้ากันได้ของ API การตอบกลับ** | รองรับ `/v1/responses` เต็มรูปแบบสำหรับ Codex และเวิร์กโฟลว์เอเจนต์ขั้นสูง | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| คุณสมบัติ | มันทำอะไร | -| ---------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**การสร้างภาพ** | `/v1/images/genes` พร้อมระบบคลาวด์และแบ็กเอนด์ในเครื่อง | -| 📐**การฝัง** | `/v1/embeddings` สำหรับการค้นหาและไปป์ไลน์ RAG | -| 🎶**การถอดเสียง** | `/v1/audio/transcriptions` — ผู้ให้บริการ 7 ราย (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), การตรวจจับภาษาอัตโนมัติ, รองรับ MP4/MP3/WAV | -| 🔊**ข้อความเป็นคำพูด** | `/v1/audio/speech` — ผู้ให้บริการ 10 ราย (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) พร้อมข้อความแสดงข้อผิดพลาดที่ถูกต้อง | -| 🎬**การสร้างวิดีโอ** | `/v1/videos/ generations` (เวิร์กโฟลว์ ComfyUI + SD WebUI) | -| 🎵**การสร้างดนตรี** | `/v1/music/ generations` (เวิร์กโฟลว์ ComfyUI) | -| 🛡️**การกลั่นกรอง** | การตรวจสอบความปลอดภัยของ `/v1/moderations` | -| 🔀**จัดอันดับ** | `/v1/rerank` สำหรับการให้คะแนนความเกี่ยวข้อง | -| 🔍**ค้นหาเว็บ**🆕 | `/v1/search` — ผู้ให้บริการ 5 ราย (Serper, Brave, Perplexity, Exa, Tavily), ฟรี 6,500+ รายการ/เดือน, เฟลโอเวอร์อัตโนมัติ, แคช | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| คุณสมบัติ | มันทำอะไร | -| -------------------------------------------- | ----------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**เซอร์กิตเบรกเกอร์** | การเดินทางต่อรุ่น/การกู้คืนด้วยการควบคุมเกณฑ์ | -| 🎯**โมเดล Endpoint-Aware** | โมเดลที่กำหนดเองประกาศจุดสิ้นสุดที่รองรับ + ​​รูปแบบ API | -| 🛡️**ฝูงต่อต้านฟ้าร้อง** | การป้องกัน Mutex + เซมาฟอร์ในการลองใหม่/ให้คะแนนเหตุการณ์ | -| 🧠**แคชความหมาย + ลายเซ็น** | การลดต้นทุน/เวลาแฝงด้วยแคชสองชั้น | -| ⚡**ขอ Idempotency** | หน้าต่างการป้องกันซ้ำ | -| 🔒**การปลอมแปลงลายนิ้วมือ TLS** | ลายนิ้วมือ TLS เหมือนเบราว์เซอร์ —**ลดการตรวจจับบอทและการตั้งค่าสถานะบัญชี** | -| 🔏**การจับคู่ลายนิ้วมือ CLI** | จับคู่ลายเซ็นคำขอ CLI ดั้งเดิม —**ลดความเสี่ยงในการแบนในขณะที่รักษา IP พร็อกซี** | -| 🌐**การกรอง IP** | การควบคุมรายการที่อนุญาต/รายการบล็อกสำหรับการปรับใช้แบบเปิดเผย | -| 📊**ขีดจำกัดอัตราที่แก้ไขได้** | ขีดจำกัดระดับโลก/ระดับผู้ให้บริการที่กำหนดค่าได้ด้วยความคงอยู่ | -| 📉**การเสื่อมถอยอย่างสง่างาม** | ทางเลือกความสามารถแบบหลายเลเยอร์ที่ปกป้องการทำงานของเกตเวย์หลัก | -| 📜**กำหนดเส้นทางการตรวจสอบ** | การติดตามการเปลี่ยนแปลงตามความแตกต่างป้องกันการดริฟท์การดำเนินงานด้วยการย้อนกลับอย่างง่าย | -| ⏳**การซิงค์สุขภาพของผู้ให้บริการ** | การตรวจสอบการหมดอายุของโทเค็นเชิงรุกทำให้เกิดการแจ้งเตือนก่อนการอนุญาตล้มเหลว | -| 🚪**ปิดการใช้งานบัญชีที่ถูกแบนโดยอัตโนมัติ** | เบรกเกอร์ปฏิบัติการปิดผนึกบัญชีโทเค็นที่ถูกบล็อกอย่างถาวรโดยอัตโนมัติ | -| 🔑**การจัดการคีย์ API + การกำหนดขอบเขต** | รักษาความปลอดภัยการออก/การหมุนเวียนคีย์ และการควบคุมโมเดล/ผู้ให้บริการ | -| 👁️**การเปิดเผยคีย์ API ที่กำหนดขอบเขต**🆕 | เลือกใช้การกู้คืนคีย์ API ผ่าน `ALLOW_API_KEY_REVEAL` | -| 🛡️**ป้องกัน `/models`** | ตัวเลือกการตรวจสอบสิทธิ์และการซ่อนผู้ให้บริการสำหรับแค็ตตาล็อกโมเดล | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| คุณสมบัติ | มันทำอะไร | -| -------------------------------- | -------------------------------------------------------------------------- | ------------- | -| 📝**คำขอ + การบันทึกพร็อกซี** | คำขอ/ตอบกลับแบบเต็มและการบันทึกพร็อกซี | -| 📉**บันทึกรายละเอียดแบบสตรีม**🆕 | สร้างสตรีมเพย์โหลด SSE ใหม่ใน UI | ได้อย่างหมดจด | -| 📋**แดชบอร์ดบันทึกแบบรวม** | มุมมองคำขอ พร็อกซี การตรวจสอบ และคอนโซลในหน้าเดียว | -| 🔍**ขอโทรมาตร** | เวลาแฝง p50/p95/p99 และคำขอการติดตาม | -| 🏥**แดชบอร์ดสุขภาพ** | เวลาทำงาน สถานะของเบรกเกอร์ การล็อก สถิติแคช | -| 💰**ติดตามต้นทุน** | การควบคุมงบประมาณและการเปิดเผยราคาต่อรุ่น | -| 📈**การแสดงภาพการวิเคราะห์** | ข้อมูลเชิงลึกการใช้งานโมเดล/ผู้ให้บริการและมุมมองแนวโน้ม | -| 🧪**กรอบการประเมินผล** | การทดสอบชุดทองพร้อมกลยุทธ์การจับคู่ที่กำหนดค่าได้ | -| 📡**การวินิจฉัยแบบเรียลไทม์**🆕 | บายพาสแคชความหมายเพื่อการทดสอบคอมโบสดที่แม่นยำ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| คุณสมบัติ | มันทำอะไร | -| ----------------------------------- | ------------------------------------------------------------------------- | --------------------- | -| 🌐**ปรับใช้ได้ทุกที่** | Localhost, VPS, Docker, สภาพแวดล้อมคลาวด์ | -| 🚇**อุโมงค์คลาวด์แฟลร์**🆕 | การรวม Quick Tunnel เพียงคลิกเดียวจากแดชบอร์ด | -| 🔑**การกรองโมเดลคีย์ API** | การตอบสนองดั้งเดิม /v1/models กรองผ่านบทบาทบริบทของผู้ถือที่ได้รับมอบหมาย | -| ⚡**บายพาสแคชอัจฉริยะ** | การวิเคราะห์พฤติกรรม TTL ที่กำหนดค่าได้และการควบคุมการดึงข้อมูลแบบบังคับ | -| 🔄**สำรอง/กู้คืน** | ขั้นตอนการส่งออก/นำเข้าและการกู้คืนความเสียหาย | -| 🧙**ตัวช่วยสร้างการเริ่มต้นใช้งาน** | การตั้งค่าที่แนะนำการใช้งานครั้งแรก | -| ????**แดชบอร์ดเครื่องมือ CLI** | ตั้งค่าเพียงคลิกเดียวสำหรับเครื่องมือเข้ารหัสยอดนิยม | -| 🎮**โมเดลสนามเด็กเล่น** | ทดสอบผู้ให้บริการ/รุ่น/ปลายทางจากแดชบอร์ด | -| 🔏**สลับลายนิ้วมือ CLI** | การจับคู่ลายนิ้วมือของแต่ละผู้ให้บริการในการตั้งค่า > ความปลอดภัย | -| 🌐**i18n (30 ภาษา)** | รองรับภาษาแดชบอร์ด + เอกสารแบบเต็มพร้อมการครอบคลุม RTL | -| 🧹**ล้างทุกรุ่น** | การล้างรายการโมเดลในคลิกเดียวในรายละเอียดผู้ให้บริการ | -| 👁️**การควบคุมแถบด้านข้าง**🆕 | ซ่อนส่วนประกอบและการผสานรวมจากการตั้งค่าลักษณะที่ปรากฏ | -| 📋**เทมเพลตปัญหา** | เทมเพลต GitHub มาตรฐานสำหรับข้อบกพร่องและฟีเจอร์ | -| 📂**ไดเรกทอรีข้อมูลที่กำหนดเอง** | แทนที่ `DATA_DIR` สำหรับตำแหน่งที่เก็บข้อมูล | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1294,105 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -เมื่อโควต้า อัตรา หรือสุขภาพล้มเหลว OmniRoute จะย้ายไปยังผู้สมัครรายถัดไปโดยอัตโนมัติโดยไม่ต้องสลับด้วยตนเอง#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A สามารถค้นพบได้ใน UI และเอกสาร (ไม่ได้ซ่อนไว้) -- API สถานะโปรโตคอลเปิดเผยข้อมูลการดำเนินงานสด (`/api/mcp/*`, `/api/a2a/*`) -- แดชบอร์ดรวมการดำเนินการสำหรับปฏิบัติการวันที่ 2 (การสลับคำสั่งผสม การรีเซ็ตเบรกเกอร์ การยกเลิกงาน)#### Translator + validation workflow +#### Protocol management that is visible and operable -พื้นที่นักแปลประกอบด้วย: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**สนามเด็กเล่น**: ขอตรวจสอบการเปลี่ยนแปลง -**ผู้ทดสอบแชท**: คำขอ/ตอบกลับแบบเต็มไปกลับ -**ม้านั่งทดสอบ**: มีหลายกรณีในการวิ่งครั้งเดียว -**Live Monitor**: มุมมองการจราจรแบบเรียลไทม์ +#### Translator + validation workflow -รวมถึงการตรวจสอบความถูกต้องของโปรโตคอลกับไคลเอนต์จริงผ่าน `npm run test:protocols:e2e` +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— การอ้างอิงเครื่องมือ การกำหนดค่า IDE และตัวอย่างไคลเอนต์ +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A Server README](src/lib/a2a/README.md)**— ทักษะ, วิธี JSON-RPC, การสตรีม และวงจรการใช้งาน## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute มีกรอบการประเมินในตัวเพื่อทดสอบคุณภาพการตอบสนองของ LLM เทียบกับชุดทอง เข้าถึงได้ผ่านทาง**Analytics → Evals**ในแดชบอร์ด### Built-in Golden Set +## 🧪 Evaluations (Evals) -"OmniRoute Golden Set" ที่โหลดไว้ล่วงหน้ามีกรณีทดสอบสำหรับ: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- คำทักทาย คณิตศาสตร์ ภูมิศาสตร์ การสร้างโค้ด -- การปฏิบัติตามรูปแบบ JSON, การแปล, การสร้างมาร์กดาวน์ -- การปฏิเสธอย่างปลอดภัย (เนื้อหาที่เป็นอันตราย) การนับ ตรรกะบูลีน### Evaluation Strategies +### Built-in Golden Set -| กลยุทธ์ | คำอธิบาย | ตัวอย่าง | -| ---------- | ------------------------------------------------------------------ | -------------------------------- | --- | -| `แน่นอน` | ผลลัพธ์จะต้องตรงกันทุกประการ | `"4"` | -| `มี' | เอาต์พุตจะต้องมีสตริงย่อย (ไม่คำนึงถึงตัวพิมพ์เล็กและตัวพิมพ์ใหญ่) | `"ปารีส"` | -| `regex` | เอาต์พุตต้องตรงกับรูปแบบ regex | `"1.*2.*3"` | -| `กำหนดเอง` | ฟังก์ชัน JS แบบกำหนดเองส่งคืนค่า true/false | `(output) => output.length > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) -<รายละเอียด> +
+🧩 MCP Setup (Model Context Protocol) -🧩 การตั้งค่า MCP (Model Context Protocol) +Start MCP transport in stdio mode: -เริ่มการขนส่ง MCP ในโหมด stdio:```bash +```bash omniroute --mcp +``` -```` +Recommended validation flow: -ขั้นตอนการตรวจสอบที่แนะนำ: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. เชื่อมต่อไคลเอนต์ MCP ของคุณผ่าน stdio -2. เรียกใช้ `omniroute_get_health` -3. เรียกใช้ `omniroute_list_combos` -4. เปิด `/dashboard/mcp` เพื่อยืนยันการเต้นของหัวใจ กิจกรรม และการตรวจสอบ +Useful APIs for automation: -API ที่มีประโยชน์สำหรับระบบอัตโนมัติ: - -- `รับ /api/mcp/สถานะ` +- `GET /api/mcp/status` - `GET /api/mcp/tools` - `GET /api/mcp/audit` -- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats` -<รายละเอียด> -🤝 การตั้งค่า A2A (Agent2Agent) + -ค้นหาตัวแทน:```bash +
+🤝 A2A Setup (Agent2Agent) + +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -ส่งงาน:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` +Manage lifecycle: -จัดการวงจรชีวิต: - -- `รับ /api/a2a/สถานะ` -- `รับ /api/a2a/งาน` -- `รับ /api/a2a/งาน/:id` +- `GET /api/a2a/status` +- `GET /api/a2a/tasks` +- `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -UI การดำเนินงาน: +Operational UI: -- `/dashboard/a2a` สำหรับการสังเกตงาน/สถานะ/สตรีมและการดำเนินการของควัน
+- `/dashboard/a2a` for task/state/stream observability and smoke actions -<รายละเอียด> -🧪 การตรวจสอบความถูกต้องของโปรโตคอลจากต้นทางถึงปลายทาง + -ตรวจสอบทั้งสองโปรโตคอลกับไคลเอนต์จริง:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -สิ่งนี้จะยืนยัน: +This verifies: -- เชื่อมต่อ/รายการ/โทรไคลเอ็นต์ MCP SDK -- การค้นพบ A2A/ส่ง/สตรีม/รับ/ยกเลิก -- ตรวจสอบข้อมูลในการตรวจสอบ MCP และ API การจัดการงาน A2A
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs -<รายละเอียด> + -💳 ผู้ให้บริการสมัครสมาชิก### Claude Code (Pro/Max) +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1405,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**เคล็ดลับสำหรับมือโปร:**ใช้ Opus สำหรับงานที่ซับซ้อน และใช้ Sonnet เพื่อความรวดเร็ว โควต้าการติดตาม OmniRoute ต่อรุ่น!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1419,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -ขณะนี้บัญชี Codex แต่ละบัญชีมีการสลับนโยบายใน 'แดชบอร์ด -> ผู้ให้บริการ': +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (เปิด/ปิด): บังคับใช้นโยบายเกณฑ์กรอบเวลา 5 ชั่วโมง -- `รายสัปดาห์` (เปิด/ปิด): บังคับใช้นโยบายเกณฑ์หน้าต่างรายสัปดาห์ -- ลักษณะการทำงานตามเกณฑ์: เมื่อหน้าต่างที่เปิดใช้งานถึงการใช้งาน >=90% บัญชีนั้นจะถูกข้ามไป -- พฤติกรรมการหมุน: OmniRoute กำหนดเส้นทางไปยังบัญชี Codex ที่มีสิทธิ์ถัดไปโดยอัตโนมัติ -- พฤติกรรมการรีเซ็ต: เมื่อผู้ให้บริการ `รีเซ็ตเมื่อเวลาผ่านไป บัญชีจะมีสิทธิ์อีกครั้งโดยอัตโนมัติ +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -สถานการณ์: +Scenarios: -- `5h ON` + `Weekly ON`: บัญชีจะถูกข้ามไปเมื่อหน้าต่างใดหน้าต่างหนึ่งถึงเกณฑ์ -- `ปิด 5 ชั่วโมง` + `เปิดรายสัปดาห์`: เฉพาะการใช้งานรายสัปดาห์เท่านั้นที่สามารถบล็อกบัญชีได้ -- `เปิด 5 ชั่วโมง` + `ปิดรายสัปดาห์`: การใช้งานเพียง 5 ชั่วโมงเท่านั้นที่สามารถบล็อกบัญชีได้ -- `resetAt` ผ่าน: บัญชีกลับเข้าสู่การหมุนอีกครั้งโดยอัตโนมัติ (ไม่มีการเปิดใช้อีกครั้งด้วยตนเอง)### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1444,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**คุ้มค่าที่สุด:**ระดับฟรีมหาศาล! ใช้สิ่งนี้ก่อนระดับที่ชำระเงิน### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1459,74 +1662,91 @@ Models:
-<รายละเอียด> +
+🔑 API Key Providers -🔑 ผู้ให้บริการคีย์ API### NVIDIA NIM (FREE developer access — 70+ models) +### NVIDIA NIM (FREE developer access — 70+ models) -1. ลงทะเบียน: [build.nvidia.com](https://build.nvidia.com) -2. รับคีย์ API ฟรี (รวมเครดิตการอนุมาน 1,000 รายการ) -3. แดชบอร์ด → เพิ่มผู้ให้บริการ → NVIDIA NIM: - - คีย์ API: `nvapi-your-key` +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**รุ่น:**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` และอีกกว่า 50 รายการ +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -**เคล็ดลับสำหรับมือโปร:**API ที่เข้ากันได้กับ OpenAI — ทำงานได้อย่างราบรื่นกับการแปลรูปแบบของ OmniRoute!### DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -1. ลงทะเบียน: [platform.deepseek.com](https://platform.deepseek.com) -2. รับรหัส API -3. แดชบอร์ด → เพิ่มผู้ให้บริการ → DeepSeek +### DeepSeek -**รุ่น:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -1. ลงทะเบียน: [console.groq.com](https://console.groq.com) -2. รับรหัส API (รวมระดับฟรี) -3. แดชบอร์ด → เพิ่มผู้ให้บริการ → Groq +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**รุ่น:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +### Groq (Free Tier Available!) -**เคล็ดลับสำหรับมือโปร:**การอนุมานที่รวดเร็วเป็นพิเศษ — ดีที่สุดสำหรับการเขียนโค้ดแบบเรียลไทม์!### OpenRouter (100+ Models) +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -1. ลงทะเบียน: [openrouter.ai](https://openrouter.ai) -2. รับรหัส API -3. แดชบอร์ด → เพิ่มผู้ให้บริการ → OpenRouter +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**รุ่น:**เข้าถึงโมเดลมากกว่า 100 โมเดลจากผู้ให้บริการรายใหญ่ทั้งหมดผ่านคีย์ API เดียว +**Pro Tip:** Ultra-fast inference — best for real-time coding! -**ลักษณะการทำงานของแดชบอร์ด:**โมเดล OpenRouter ได้รับการจัดการจาก**รุ่นที่มีจำหน่าย**เพิ่ม นำเข้า และซิงค์อัตโนมัติด้วยตนเองทั้งหมดจะอัปเดตรายการเดียวกัน
+### OpenRouter (100+ Models) -<รายละเอียด> +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -💰 ผู้ให้บริการราคาถูก (สำรอง)### GLM-4.7 (Daily reset, $0.6/1M) +**Models:** Access 100+ models from all major providers through a single API key. -1. ลงทะเบียน: [Zhipu AI](https://open.bigmodel.cn/) -2. รับคีย์ API จาก Coding Plan -3. แดชบอร์ด → เพิ่มคีย์ API: - - ผู้ให้บริการ: `glm` - - คีย์ API: `คีย์ของคุณ` +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -**ใช้:**`glm/glm-4.7` + -**เคล็ดลับสำหรับมือโปร:**แผนการเขียนโค้ดเสนอโควต้า 3 เท่าในราคา 1/7! รีเซ็ตทุกวัน 10.00 น.### MiniMax M2.1 (5h reset, $0.20/1M) +
+💰 Cheap Providers (Backup) -1. ลงทะเบียน: [MiniMax](https://www.minimax.io/) -2. รับรหัส API -3. แดชบอร์ด → เพิ่มคีย์ API +### GLM-4.7 (Daily reset, $0.6/1M) -**การใช้งาน:**`minimax/MiniMax-M2.1` +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**เคล็ดลับสำหรับมือโปร:**ตัวเลือกที่ถูกที่สุดสำหรับบริบทที่ยาว (โทเค็น 1M)!### Kimi K2 ($9/month flat) +**Use:** `glm/glm-4.7` -1. สมัครสมาชิก: [Moonshot AI](https://platform.moonshot.ai/) -2. รับรหัส API -3. แดชบอร์ด → เพิ่มคีย์ API +**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. -**ใช้:**`kimi/kimi-latest` +### MiniMax M2.1 (5h reset, $0.20/1M) -**เคล็ดลับสำหรับมือโปร:**แก้ไข $9/เดือนสำหรับโทเค็น 10M = $0.90/ต้นทุนที่แท้จริง 1M!
+1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key -<รายละเอียด> +**Use:** `minimax/MiniMax-M2.1` -🆓 ผู้ให้บริการฟรี (การสำรองข้อมูลฉุกเฉิน)### Qoder (5 FREE models via OAuth) +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1567,9 +1787,10 @@ Models:
-<รายละเอียด> +
+🎨 Create Combos -🎨 สร้างคอมโบ### Example 1: Maximize Subscription → Cheap Backup +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1597,9 +1818,10 @@ Cost: $0 forever!
-<รายละเอียด> +
+🔧 CLI Integration -🏽 บูรณาการ CLI### Cursor IDE +### Cursor IDE ``` Settings → Models → Advanced: @@ -1610,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -ใช้หน้า**เครื่องมือ CLI**ในแดชบอร์ดเพื่อกำหนดค่าด้วยคลิกเดียว หรือแก้ไข `~/.claude/settings.json` ด้วยตนเอง### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1621,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**ตัวเลือก 1 — แดชบอร์ด (แนะนำ):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**ตัวเลือก 2 — กำหนดเอง:**แก้ไข `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1638,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **หมายเหตุ:**OpenClaw ใช้งานได้กับ OmniRoute ในพื้นที่เท่านั้น ใช้ `127.0.0.1` แทน `localhost` เพื่อหลีกเลี่ยงปัญหาความละเอียดของ IPv6### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1652,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**ขั้นตอนที่ 1:**เพิ่ม OmniRoute เป็นผู้ให้บริการแบบกำหนดเอง:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**ขั้นตอนที่ 2:**สร้าง/แก้ไข `opencode.json` ในรูทโปรเจ็กต์ของคุณ:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1678,117 +1909,130 @@ opencode } } } -```` +``` -**ขั้นตอนที่ 3:**เลือกโมเดลใน OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**เคล็ดลับ:**เพิ่มโมเดลใดๆ ที่มีอยู่ในปลายทาง OmniRoute `/v1/models` ของคุณไปยังส่วน `models` ใช้รูปแบบ `ผู้ให้บริการ/รหัสรุ่น` จากแดชบอร์ด OmniRoute ของคุณ
+ --- ## การแก้ไขปัญหา -<รายละเอียด> -คลิกเพื่อขยายคำแนะนำการแก้ปัญหา +
+Click to expand troubleshooting guide -**"โมเดลภาษาไม่ได้ระบุข้อความ"** +**"Language model did not provide messages"** -- โควต้าผู้ให้บริการหมด → ตรวจสอบตัวติดตามโควต้าแดชบอร์ด -- วิธีแก้ไข: ใช้ทางเลือกแบบคอมโบหรือเปลี่ยนไปใช้ระดับที่ถูกกว่า +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**จำกัดอัตรา** +**Rate limiting** -- โควต้าการสมัครสมาชิกหมด → สำรองไปที่ GLM/MiniMax -- เพิ่มคอมโบ: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**โทเค็น OAuth หมดอายุแล้ว** +**OAuth token expired** -- รีเฟรชอัตโนมัติโดย OmniRoute -- หากปัญหายังคงอยู่: แดชบอร์ด → ผู้ให้บริการ → เชื่อมต่อใหม่ +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**ค่าใช้จ่ายสูง** +**High costs** -- ตรวจสอบสถิติการใช้งานในแดชบอร์ด → ต้นทุน -- เปลี่ยนโมเดลหลักเป็น GLM/MiniMax -- ใช้ Free Tier (Gemini CLI, Qoder) สำหรับงานที่ไม่สำคัญ +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**พอร์ตแดชบอร์ด/API ไม่ถูกต้อง** +**Dashboard/API ports are wrong** -- `PORT` คือพอร์ตฐานมาตรฐาน (และพอร์ต API เป็นค่าเริ่มต้น) -- `API_PORT` จะแทนที่เฉพาะ Listener API ที่เข้ากันได้กับ OpenAI เท่านั้น -- `DASHBOARD_PORT` จะแทนที่เฉพาะตัวฟังแดชบอร์ด/Next.js -- ตั้งค่า `NEXT_PUBLIC_BASE_URL` เป็นแดชบอร์ด/URL สาธารณะของคุณ (สำหรับการเรียกกลับ OAuth) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**ข้อผิดพลาดในการซิงค์คลาวด์** +**Cloud sync errors** -- ตรวจสอบ `BASE_URL` ชี้ไปที่อินสแตนซ์ที่ทำงานอยู่ของคุณ -- ตรวจสอบ `CLOUD_URL` ชี้ไปยังจุดปลายทางคลาวด์ที่คุณคาดหวัง -- รักษาค่า `NEXT_PUBLIC_*` ให้สอดคล้องกับค่าฝั่งเซิร์ฟเวอร์ +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**เข้าสู่ระบบครั้งแรกใช้งานไม่ได้** +**First login not working** -- ตรวจสอบ `INITIAL_PASSWORD` ใน `.env` -- หากไม่ได้ตั้งค่า รหัสผ่านสำรองคือ `123456` +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**ไม่มีบันทึกคำขอ** +**No request logs** -- อาร์ติแฟกต์คำขอถูกเขียนไปที่ `DATA_DIR/call_logs/` เป็นไฟล์ JSON หนึ่งไฟล์ต่อคำขอ -- เปิดใช้งานการจับภาพไปป์ไลน์จากแดชบอร์ด → บันทึก → ขอบันทึก หากคุณต้องการเพย์โหลดต่อขั้นตอนโดยละเอียด -- ตั้งค่า `APP_LOG_TO_FILE=true` หากคุณต้องการบันทึกคอนโซลแอปพลิเคชันใน `logs/application/app.log` -- ปรับ `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` และ `CALL_LOG_MAX_ENTRIES` ตามต้องการ +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**การทดสอบการเชื่อมต่อแสดงว่า "ไม่ถูกต้อง" สำหรับผู้ให้บริการที่รองรับ OpenAI** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- ผู้ให้บริการหลายรายไม่เปิดเผยจุดสิ้นสุด `/models` -- OmniRoute v1.0.6+ มีการตรวจสอบทางเลือกผ่านการแชทให้เสร็จสิ้น -- ตรวจสอบให้แน่ใจว่า URL พื้นฐานมีคำต่อท้าย `/v1`### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ สำคัญสำหรับผู้ใช้ที่ใช้ OmniRoute บน VPS, Docker หรือเซิร์ฟเวอร์ระยะไกลใดๆ**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -ผู้ให้บริการ**Antigravity**และ**Gemini CLI**ใช้**Google OAuth 2.0**Google กำหนดให้ใช้ `redirect_uri` ในโฟลว์ OAuth เพื่อให้ตรงกับหนึ่งใน URI ที่ลงทะเบียนล่วงหน้าใน Google Cloud Console ของแอป +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -ข้อมูลรับรอง OAuth ที่รวมอยู่ใน OmniRoute ได้รับการลงทะเบียน**สำหรับ `localhost` เท่านั้น**เมื่อคุณเข้าถึง OmniRoute บนเซิร์ฟเวอร์ระยะไกล (เช่น `https://omniroute.myserver.com`) Google จะปฏิเสธการตรวจสอบสิทธิ์ด้วย:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -คุณต้องสร้าง**OAuth 2.0 Client ID**ใน Google Cloud Console ด้วย URI ของเซิร์ฟเวอร์ของคุณ#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. เปิดคอนโซล Google Cloud** +#### Step-by-step -ไปที่: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. สร้างรหัสไคลเอ็นต์ OAuth 2.0 ใหม่** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- คลิก**"+ สร้างข้อมูลรับรอง"**→**"รหัสไคลเอ็นต์ OAuth"** -- ประเภทแอปพลิเคชัน:**"แอปพลิเคชันเว็บ"** -- ชื่อ: อะไรก็ได้ที่คุณชอบ (เช่น `OmniRoute Remote`) +**2. Create a new OAuth 2.0 Client ID** -**3. เพิ่ม URI การเปลี่ยนเส้นทางที่ได้รับอนุญาต** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -ในช่อง**"URI การเปลี่ยนเส้นทางที่ได้รับอนุญาต"**ให้เพิ่ม:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> แทนที่ `your-server.com` ด้วยโดเมนหรือ IP ของเซิร์ฟเวอร์ของคุณ (รวมพอร์ตหากจำเป็น เช่น `http://45.33.32.156:20128/callback`) +**4. Save and copy the credentials** -**4. บันทึกและคัดลอกข้อมูลรับรอง** +After creating, Google will show the **Client ID** and **Client Secret**. -หลังจากสร้างแล้ว Google จะแสดง**รหัสไคลเอ็นต์**และ**รหัสลับไคลเอ็นต์** +**5. Set environment variables** -**5. ตั้งค่าตัวแปรสภาพแวดล้อม** +In your `.env` (or Docker environment variables): -ใน `.env` ของคุณ (หรือตัวแปรสภาพแวดล้อม Docker):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1797,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. รีสตาร์ท OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. ลองเชื่อมต่ออีกครั้ง** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -แดชบอร์ด → ผู้ให้บริการ → Antigravity (หรือ Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -ตอนนี้ Google จะเปลี่ยนเส้นทางไปยัง `https://your-server.com/callback` อย่างถูกต้อง--- +--- #### Temporary workaround (without custom credentials) -หากคุณไม่ต้องการตั้งค่าข้อมูลรับรองของคุณเองในขณะนี้ คุณยังคงใช้**ขั้นตอน URL ด้วยตนเอง**ได้: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute เปิด URL การอนุญาตของ Google -2. หลังจากอนุญาตแล้ว Google จะพยายามเปลี่ยนเส้นทางไปยัง `localhost` (ซึ่งล้มเหลวบนเซิร์ฟเวอร์ระยะไกล) -3.**คัดลอก URL แบบเต็ม**จากแถบที่อยู่ของเบราว์เซอร์ของคุณ (แม้ว่าหน้าเว็บจะไม่โหลดก็ตาม) -4. วาง URL นั้นลงในฟิลด์ที่แสดงในโมดอลการเชื่อมต่อ OmniRoute -5. คลิก**"เชื่อมต่อ"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> ใช้งานได้เนื่องจากรหัสการให้สิทธิ์ใน URL นั้นถูกต้อง ไม่ว่าหน้าการเปลี่ยนเส้นทางจะโหลดแล้วก็ตาม--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. -<รายละเอียด> -🇧🇷 Versão em Português#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -ระบบปฏิบัติการ**Antigravity**และ**Gemini CLI**ใช้**Google OAuth 2.0**สำหรับการรับรองความถูกต้อง O Google ต้องการ `redirect_uri` usada no fluxo OAuth seja**exatamente**uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. +
+🇧🇷 Versão em Português -ตามที่รับรอง OAuth ไม่รวม OmniRoute estão cadastradas**apenas para `localhost`**เรียกใช้ OmniRoute บนเซิร์ฟเวอร์ระยะไกล (เช่น: `https://omniroute.meuservidor.com`) หรือ Google ตรวจสอบความถูกต้องของ com:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -อยู่ในนั้นโดยตรง**OAuth 2.0 Client ID**ไม่มี Google Cloud Console พร้อม URI สำหรับเซิร์ฟเวอร์#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. เข้าถึง Google Cloud Console** +#### Passo a passo -อับรา: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Acesse o Google Cloud Console** -**2. ฉันเพิ่งค้นพบ OAuth 2.0 Client ID** +Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- คลิก em**"+ สร้างข้อมูลรับรอง"**→**"รหัสไคลเอ็นต์ OAuth"** -- เคล็ดลับการใช้งาน:**"แอปพลิเคชันเว็บ"** -- ชื่อ: ชื่อ escolha qualquer (เช่น: `OmniRoute Remote`) +**2. Crie um novo OAuth 2.0 Client ID** -**3. Adicione เป็น URI การเปลี่ยนเส้นทางที่ได้รับอนุญาต** +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -ไม่มีค่าย**"URI การเปลี่ยนเส้นทางที่ได้รับอนุญาต"**ผู้สนับสนุน:``` +**3. Adicione as Authorized Redirect URIs** + +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`) +**4. Salve e copie as credenciais** -**4. Salve e copy as credenciais** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -อ้างอิงถึง Google มากที่สุด o**Client ID**e o**Client Secret** +**5. Configure as variáveis de ambiente** -**5. กำหนดค่าเป็น variáveis de Ambiente** +No seu `.env` (ou nas variáveis de ambiente do Docker): -ไม่มี `.env` (หรือ nas variáveis de Ambiente do Docker):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1876,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. ย้อนกลับไปสู่ ​​OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute +``` -```` +**7. Tente conectar novamente** -**7. เต็นท์ คอนเนกตาร์ โนวาเมนเต** +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -แดชบอร์ด → ผู้ให้บริการ → Antigravity (หรือ Gemini CLI) → OAuth +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. -ภาพรวมของ Google การแก้ไขสำหรับ `https://seu-servidor.com/callback` และการรับรองความถูกต้อง--- +--- #### Workaround temporário (sem configurar credenciais próprias) -Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo**manual de URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. O OmniRoute ย่อ URL อัตโนมัติของ Google -2. ใช้เป็นคำสั่งอัตโนมัติ หรือ Google กำหนดแนวทางใหม่สำหรับ `localhost` (que falha no servidor remoto) -3.**คัดลอก URL ที่สมบูรณ์**da barra de endereço do seu browser (mesmo que a página não carregue) +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) 4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute -5. คลิกที่นี่**"เชื่อมต่อ"** +5. Clique em **"Connect"** -> วิธีแก้ปัญหาเบื้องต้นคือทำการเปลี่ยนเส้นทางโดยอัตโนมัติและทำการเปลี่ยนเส้นทางโดยอัตโนมัติ
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1914,64 +2171,73 @@ Se não quiser criar credenciais próprias agora, ainda é possível usar o flux ## 🛠️ Tech Stack -<รายละเอียด> -คลิกเพื่อขยายรายละเอียด Tech Stack +
+Click to expand tech stack details --**รันไทม์**: Node.js 18–22 LTS (⚠️ Node.js 24+ ไม่ได้รับการสนับสนุน**— ไบนารีเนทิฟ `better-sqlite3` เข้ากันไม่ได้) --**ภาษา**: TypeScript 5.9 —**TypeScript 100%**ทั่วทั้ง `src/` และ `open-sse/` (ไม่มี `ใดๆ` ในโมดูลหลักตั้งแต่เวอร์ชัน 2.0) --**เฟรมเวิร์ก**: Next.js 16 + React 19 + Tailwind CSS 4 --**ฐานข้อมูล**: LowDB (JSON) + SQLite (สถานะโดเมน + บันทึกพร็อกซี + การตรวจสอบ MCP + การตัดสินใจกำหนดเส้นทาง) --**Schemas**: Zod (การตรวจสอบ I/O ของเครื่องมือ MCP, สัญญา API) --**โปรโตคอล**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**การสตรีม**: เหตุการณ์ที่เซิร์ฟเวอร์ส่ง (SSE) --**การตรวจสอบสิทธิ์**: OAuth 2.0 (PKCE) + JWT + คีย์ API + การอนุญาตขอบเขต MCP --**การทดสอบ**: ตัวรันการทดสอบ Node.js + Vitest (การทดสอบมากกว่า 900 รายการ รวมถึงหน่วย, การรวม, E2E) --**CI/CD**: การดำเนินการ GitHub (เผยแพร่ npm อัตโนมัติ + Docker Hub เมื่อวางจำหน่าย) --**เว็บไซต์**: [omniroute.online](https://omniroute.online) --**แพ็คเกจ**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**นักเทียบท่า**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**ความยืดหยุ่น**: เซอร์กิตเบรกเกอร์, การถอยกลับแบบเอ็กซ์โปเนนเชียล, การป้องกันฝูงฟ้าผ่า, การปลอมแปลง TLS, การรักษาตัวเองแบบคอมโบอัตโนมัติ
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## เอกสาร -| เอกสาร | คำอธิบาย | -| -------------------------------------------------- | --------------------------------------------------- | -| [คู่มือผู้ใช้](docs/USER_GUIDE.md) | ผู้ให้บริการ, คอมโบ, การรวม CLI, การปรับใช้ | -| [การอ้างอิง API](docs/API_REFERENCE.md) | จุดสิ้นสุดทั้งหมดพร้อมตัวอย่าง | -| [เซิร์ฟเวอร์ MCP](open-sse/mcp-server/README.md) | เครื่องมือ MCP 16 รายการ, การกำหนดค่า IDE, ไคลเอนต์ Python/TS/Go | -| [เซิร์ฟเวอร์ A2A](src/lib/a2a/README.md) | โปรโตคอล JSON-RPC 2.0, ทักษะ, การสตรีม, การจัดการงาน | -| [เครื่องยนต์คอมโบอัตโนมัติ](docs/auto-combo.md) | การให้คะแนน 6 ปัจจัย แพ็กโหมด การรักษาตัวเอง | -| [การแก้ไขปัญหา](docs/TROUBLESHOOTING.md) | ปัญหาและแนวทางแก้ไขทั่วไป | -| [สถาปัตยกรรม](docs/ARCHITECTURE.md) | สถาปัตยกรรมระบบและภายใน | -| [มีส่วนร่วม](CONTRIBUTING.md) | การตั้งค่าและแนวทางการพัฒนา | -| [ข้อมูลจำเพาะของ OpenAPI](docs/openapi.yaml) | ข้อมูลจำเพาะของ OpenAPI 3.0 | -| [นโยบายความปลอดภัย](SECURITY.md) | การรายงานช่องโหว่และแนวปฏิบัติด้านความปลอดภัย | -| [การปรับใช้ VM](docs/VM_DEPLOYMENT_GUIDE.md) | คู่มือฉบับสมบูรณ์: VM + nginx + การตั้งค่า Cloudflare | -| [แกลเลอรีคุณลักษณะ](docs/FEATURES.md) | ทัวร์ชมแดชบอร์ดภาพพร้อมภาพหน้าจอ | -| [รายการตรวจสอบการเผยแพร่](docs/RELEASE_CHECKLIST.md) | ขั้นตอนการตรวจสอบก่อนเผยแพร่ |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute มี**คุณสมบัติมากกว่า 210 รายการ**ที่วางแผนไว้**ในขั้นตอนการพัฒนาหลายขั้นตอน นี่คือประเด็นสำคัญ: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| หมวดหมู่ | คุณสมบัติที่วางแผนไว้ | ไฮไลท์ | -| --------------------------------- | ---------------- | -------------------------------------------------------------------------------------- | -| 🧠**การกำหนดเส้นทางและความฉลาด**| 25+ | การกำหนดเส้นทางที่มีความหน่วงต่ำที่สุด, การกำหนดเส้นทางตามแท็ก, โควต้า preflight, การเลือกบัญชี P2C | -| 🔒**ความปลอดภัยและการปฏิบัติตามข้อกำหนด**| 20+ | การเสริมความแข็งแกร่งของ SSRF, การปิดบังข้อมูลรับรอง, ขีดจำกัดอัตราต่อจุดสิ้นสุด, การกำหนดขอบเขตคีย์การจัดการ | -| 📊**ความสามารถในการสังเกต**| 15+ | การรวม OpenTelemetry การตรวจสอบโควต้าแบบเรียลไทม์ การติดตามต้นทุนต่อรุ่น | -| 🔄**การบูรณาการของผู้ให้บริการ**| 20+ | การลงทะเบียนโมเดลแบบไดนามิก, คูลดาวน์ของผู้ให้บริการ, Codex หลายบัญชี, การแยกวิเคราะห์โควต้า Copilot -| ⚡**ประสิทธิภาพ**| 15+ | เลเยอร์แคชคู่, แคชพร้อมท์, แคชการตอบสนอง, การสตรีมแบบ Keepalive, ชุด API | -| 🌐**ระบบนิเวศ**| 10+ | WebSocket API, กำหนดค่า hot-reload, การจัดเก็บ config แบบกระจาย, โหมดเชิงพาณิชย์ |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**การรวม OpenCode**— รองรับผู้ให้บริการเนทีฟสำหรับ IDE การเข้ารหัส OpenCode AI -- 🔗**การบูรณาการ TRAE**— รองรับกรอบการพัฒนา TRAE AI อย่างเต็มที่ -- 📦**Batch API**— การประมวลผลแบตช์แบบอะซิงโครนัสสำหรับคำขอจำนวนมาก -- 🎯**การกำหนดเส้นทางตามแท็ก**— คำขอกำหนดเส้นทางตามแท็กที่กำหนดเองและข้อมูลเมตา -- 💰**กลยุทธ์ต้นทุนต่ำสุด**— เลือกผู้ให้บริการที่ถูกที่สุดโดยอัตโนมัติ +### 🔜 Coming Soon -> 📝 ข้อมูลจำเพาะคุณสมบัติแบบเต็มมีอยู่ใน [`docs/new-features/`](docs/new-features/) (217 ข้อมูลจำเพาะโดยละเอียด)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1979,18 +2245,20 @@ OmniRoute มี**คุณสมบัติมากกว่า 210 ราย ### How to Contribute -1. แยกพื้นที่เก็บข้อมูล -2. สร้างสาขาฟีเจอร์ของคุณ (`git checkout -b Feature/amazing-feature`) -3. ยอมรับการเปลี่ยนแปลงของคุณ (`git commit -m 'เพิ่มฟีเจอร์ที่น่าทึ่ง'') -4. พุชไปที่สาขา (`git push origin features/amazing-feature`) -5. เปิดคำขอดึง +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -ดู [CONTRIBUTING.md](CONTRIBUTING.md) สำหรับคำแนะนำโดยละเอียด### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -2002,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -ขอขอบคุณเป็นพิเศษสำหรับ**[9router](https://github.com/decolua/9router)**โดย**[decolua](https://github.com/decolua)**— โปรเจ็กต์ดั้งเดิมที่เป็นแรงบันดาลใจให้กับ Fork นี้ OmniRoute สร้างบนรากฐานอันน่าทึ่งดังกล่าวด้วยคุณสมบัติเพิ่มเติม API แบบหลายรูปแบบ และการเขียน TypeScript ใหม่ทั้งหมด +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -ขอขอบคุณเป็นพิเศษสำหรับ**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— การใช้งาน Go ดั้งเดิมที่เป็นแรงบันดาลใจให้กับพอร์ต JavaScript นี้--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## สิทธิ์การใช้งาน -ใบอนุญาต MIT - ดู [ใบอนุญาต](ใบอนุญาต) สำหรับรายละเอียด--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/th/docs/ARCHITECTURE.md b/docs/i18n/th/docs/ARCHITECTURE.md index e6abe9382b..fba6b1b965 100644 --- a/docs/i18n/th/docs/ARCHITECTURE.md +++ b/docs/i18n/th/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_อัพเดตล่าสุด: 2026-03-28_## Executive Summary -OmniRoute เป็นเกตเวย์การกำหนดเส้นทาง AI ในพื้นที่และแดชบอร์ดที่สร้างขึ้นบน Next.js -โดยให้จุดสิ้นสุดที่เข้ากันได้กับ OpenAI จุดเดียว (`/v1/*`) และกำหนดเส้นทางการรับส่งข้อมูลผ่านผู้ให้บริการอัปสตรีมหลายรายพร้อมการแปล ทางเลือกสำรอง การรีเฟรชโทเค็น และการติดตามการใช้งาน -ความสามารถหลัก: +_Last updated: 2026-03-28_ -- พื้นผิว API ที่เข้ากันได้กับ OpenAI สำหรับ CLI/เครื่องมือ (ผู้ให้บริการ 28 ราย) -- การแปลคำขอ/ตอบกลับในรูปแบบต่างๆ ของผู้ให้บริการ -- ทางเลือกคำสั่งผสมโมเดล (ลำดับหลายรุ่น) -- ทางเลือกระดับบัญชี (หลายบัญชีต่อผู้ให้บริการ) -- การจัดการการเชื่อมต่อผู้ให้บริการ OAuth + API-key -- การสร้างการฝังผ่าน `/v1/embeddings` (ผู้ให้บริการ 6 ราย, 9 โมเดล) -- การสร้างภาพผ่าน `/v1/images/ generations` (ผู้ให้บริการ 4 ราย, 9 รุ่น) -- คิดการแยกวิเคราะห์แท็ก (`...`) สำหรับโมเดลการให้เหตุผล -- การตอบสนองการฆ่าเชื้อสำหรับความเข้ากันได้ของ OpenAI SDK ที่เข้มงวด -- การปรับบทบาทให้เป็นมาตรฐาน (ผู้พัฒนา → ระบบ, ระบบ → ผู้ใช้) เพื่อความเข้ากันได้ระหว่างผู้ให้บริการ -- การแปลงเอาต์พุตที่มีโครงสร้าง (json_schema → Gemini responseSchema) -- ความคงอยู่ในท้องถิ่นสำหรับผู้ให้บริการ คีย์ นามแฝง คอมโบ การตั้งค่า การกำหนดราคา -- การติดตามการใช้งาน/ต้นทุน และขอบันทึก -- ตัวเลือกการซิงค์บนคลาวด์สำหรับการซิงค์หลายอุปกรณ์/สถานะ -- รายการที่อนุญาต/รายการบล็อก IP สำหรับการควบคุมการเข้าถึง API -- คิดการจัดการงบประมาณ (ส่งผ่าน/อัตโนมัติ/กำหนดเอง/ปรับเปลี่ยน) -- ระบบฉีดพร้อมท์ทั่วโลก -- การติดตามเซสชันและการพิมพ์ลายนิ้วมือ -- การจำกัดอัตราการปรับปรุงต่อบัญชีด้วยโปรไฟล์เฉพาะของผู้ให้บริการ -- รูปแบบเซอร์กิตเบรกเกอร์เพื่อความยืดหยุ่นของผู้ให้บริการ -- ป้องกันฝูงฟ้าผ่าพร้อมระบบล็อค mutex -- แคชการขจัดข้อมูลซ้ำซ้อนของคำขอตามลายเซ็น -- เลเยอร์โดเมน: ความพร้อมใช้งานของโมเดล กฎต้นทุน นโยบายทางเลือก นโยบายการล็อก -- การคงอยู่ของสถานะโดเมน (แคชการเขียนผ่าน SQLite สำหรับทางเลือกสำรอง งบประมาณ การล็อคเอาต์ เซอร์กิตเบรกเกอร์) -- กลไกนโยบายสำหรับการประเมินคำขอแบบรวมศูนย์ (ล็อค → งบประมาณ → ทางเลือก) -- ขอการตรวจวัดทางไกลด้วยการรวมเวลาแฝง p50/p95/p99 -- Correlation ID (X-Request-Id) สำหรับการติดตามจากต้นทางถึงปลายทาง -- การบันทึกการตรวจสอบการปฏิบัติตามข้อกำหนดโดยเลือกไม่ใช้ต่อคีย์ API -- กรอบการประเมินสำหรับการประกันคุณภาพ LLM -- แดชบอร์ด UI ความยืดหยุ่นพร้อมสถานะเบรกเกอร์แบบเรียลไทม์ -- ผู้ให้บริการ OAuth แบบโมดูลาร์ (12 โมดูลแต่ละโมดูลภายใต้ `src/lib/oauth/providers/`) +## Executive Summary -โมเดลรันไทม์หลัก: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- เส้นทางแอป Next.js ภายใต้ `src/app/api/*` ใช้ทั้ง API แดชบอร์ดและ API ที่เข้ากันได้ -- SSE/routing core ที่ใช้ร่วมกันใน `src/sse/*` + `open-sse/*` จัดการการดำเนินการของผู้ให้บริการ การแปล การสตรีม ทางเลือกสำรอง และการใช้งาน## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- รันไทม์เกตเวย์ท้องถิ่น -- API การจัดการแดชบอร์ด -- การรับรองความถูกต้องของผู้ให้บริการและการรีเฟรชโทเค็น -- ขอการแปลและการสตรีม SSE -- สภาพท้องถิ่น + ความคงทนในการใช้งาน -- การประสานการซิงค์บนคลาวด์เสริม### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- การใช้งานบริการคลาวด์เบื้องหลัง `NEXT_PUBLIC_CLOUD_URL` -- SLA ของผู้ให้บริการ/ระนาบควบคุมอยู่นอกกระบวนการท้องถิ่น -- ไบนารี CLI ภายนอกเอง (Claude CLI, Codex CLI ฯลฯ )## Dashboard Surface (Current) +### Out of Scope -หน้าหลักภายใต้ `src/app/(dashboard)/dashboard/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — เริ่มต้นอย่างรวดเร็ว + ภาพรวมของผู้ให้บริการ -- `/dashboard/endpoint` — พร็อกซีปลายทาง + MCP + A2A + แท็บปลายทาง API -- `/dashboard/providers` — การเชื่อมต่อและข้อมูลประจำตัวของผู้ให้บริการ -- `/dashboard/combos` — กลยุทธ์คอมโบ เทมเพลต กฎการกำหนดเส้นทางโมเดล -- `/dashboard/costs` — การรวมต้นทุนและการมองเห็นราคา -- `/dashboard/analytics` — การวิเคราะห์และการประเมินผลการใช้งาน -- `/dashboard/limits` — การควบคุมโควต้า/อัตรา -- `/dashboard/cli-tools` — การเริ่มต้นใช้งาน CLI, การตรวจจับรันไทม์, การสร้างการกำหนดค่า -- `/dashboard/agents` — ตรวจพบตัวแทน ACP + การลงทะเบียนตัวแทนแบบกำหนดเอง -- `/dashboard/media` — รูปภาพ/วิดีโอ/สนามเด็กเล่นเพลง -- `/dashboard/search-tools` — การทดสอบและประวัติของผู้ให้บริการค้นหา -- `/dashboard/health` — สถานะการออนไลน์ เซอร์กิตเบรกเกอร์ ขีดจำกัดอัตรา -- `/dashboard/logs` — บันทึกคำขอ/พร็อกซี/การตรวจสอบ/คอนโซล -- `/dashboard/settings` — แท็บการตั้งค่าระบบ (ทั่วไป การกำหนดเส้นทาง ค่าเริ่มต้นคอมโบ ฯลฯ) -- `/dashboard/api-manager` — วงจรการใช้งานคีย์ API และการอนุญาตโมเดล## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -ไดเรกทอรีหลัก: +Main directories: -- `src/app/api/v1/*` และ `src/app/api/v1beta/*` สำหรับ API ที่เข้ากันได้ -- `src/app/api/*` สำหรับ API การจัดการ/การกำหนดค่า -- ถัดไปเขียนใหม่ในแมป `next.config.mjs` `/v1/*` เป็น `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -เส้นทางความเข้ากันได้ที่สำคัญ: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` — รวมโมเดลที่กำหนดเองด้วย `custom: true` -- `src/app/api/v1/embeddings/route.ts` — การสร้างการฝัง (ผู้ให้บริการ 6 ราย) -- `src/app/api/v1/images/ generations/route.ts` — การสร้างภาพ (ผู้ให้บริการ 4+ รายรวม Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — แชทเฉพาะต่อผู้ให้บริการ -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — การฝังต่อผู้ให้บริการโดยเฉพาะ -- `src/app/api/v1/providers/[provider]/images/ generations/route.ts` — อิมเมจต่อผู้ให้บริการโดยเฉพาะ +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` - `src/app/api/v1beta/models/[...path]/route.ts` -โดเมนการจัดการ: +Management domains: -- การรับรองความถูกต้อง/การตั้งค่า: `src/app/api/auth/*`, `src/app/api/settings/*` -- ผู้ให้บริการ/การเชื่อมต่อ: `src/app/api/providers*` -- โหนดผู้ให้บริการ: `src/app/api/provider-nodes*` -- โมเดลที่กำหนดเอง: `src/app/api/provider-models` (GET/POST/DELETE) -- แคตตาล็อกโมเดล: `src/app/api/models/route.ts` (GET) -- การกำหนดค่าพร็อกซี: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- คีย์/นามแฝง/คอมโบ/การกำหนดราคา: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- การใช้งาน: `src/app/api/usage/*` -- ซิงค์/คลาวด์: `src/app/api/sync/*`, `src/app/api/cloud/*` -- ผู้ช่วยเครื่องมือ CLI: `src/app/api/cli-tools/*` -- ตัวกรอง IP: `src/app/api/settings/ip-filter` (GET/PUT) -- งบประมาณการคิด: `src/app/api/settings/thinking-budget` (GET/PUT) -- พร้อมท์ระบบ: `src/app/api/settings/system-prompt` (GET/PUT) -- เซสชัน: `src/app/api/sessions` (GET) -- ขีดจำกัดอัตรา: `src/app/api/rate-limits` (GET) -- ความยืดหยุ่น: `src/app/api/resilience` (GET/PATCH) — โปรไฟล์ผู้ให้บริการ, เซอร์กิตเบรกเกอร์, สถานะขีดจำกัดอัตรา -- รีเซ็ตความยืดหยุ่น: `src/app/api/resilience/reset` (POST) — รีเซ็ตเบรกเกอร์ + คูลดาวน์ -- สถิติแคช: `src/app/api/cache/stats` (GET/DELETE) -- ความพร้อมใช้งานของโมเดล: `src/app/api/models/availability` (GET/POST) -- การวัดและส่งข้อมูลทางไกล: `src/app/api/telemetry/summary` (GET) -- งบประมาณ: `src/app/api/usage/budget` (GET/POST) -- เชนทางเลือก: `src/app/api/fallback/chains` (GET/POST/DELETE) -- การตรวจสอบการปฏิบัติตามข้อกำหนด: `src/app/api/compliance/audit-log` (GET) +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) - Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- นโยบาย: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Policies: `src/app/api/policies` (GET/POST) -โมดูลการไหลหลัก: +## 2) SSE + Translation Core -- รายการ: `src/sse/handlers/chat.ts` -- การประสานหลัก: `open-sse/handlers/chatCore.ts` -- อะแดปเตอร์การดำเนินการของผู้ให้บริการ: `open-sse/executors/*` -- รูปแบบการตรวจจับ/การกำหนดค่าผู้ให้บริการ: `open-sse/services/provider.ts` -- โมเดลแยกวิเคราะห์ / แก้ไข: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- ตรรกะทางเลือกของบัญชี: `open-sse/services/accountFallback.ts` -- รีจิสทรีการแปล: `open-sse/translator/index.ts` -- การแปลงสตรีม: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- การแยกการใช้งาน/การทำให้เป็นมาตรฐาน: `open-sse/utils/usageTracking.ts` -- คิดว่าตัวแยกวิเคราะห์แท็ก: `open-sse/utils/thinkTagParser.ts` -- ตัวจัดการการฝัง: `open-sse/handlers/embeddings.ts` -- การลงทะเบียนผู้ให้บริการการฝัง: `open-sse/config/embeddingRegistry.ts` -- ตัวจัดการการสร้างอิมเมจ: `open-sse/handlers/imageGeneration.ts` -- รีจิสทรีของผู้ให้บริการอิมเมจ: `open-sse/config/imageRegistry.ts` -- การตอบสนองการฆ่าเชื้อ: `open-sse/handlers/responseSanitizer.ts` -- การทำให้บทบาทเป็นมาตรฐาน: `open-sse/services/roleNormalizer.ts` +Main flow modules: -บริการ (ตรรกะทางธุรกิจ): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- การเลือกบัญชี/การให้คะแนน: `open-sse/services/accountSelector.ts` -- การจัดการวงจรชีวิตบริบท: `open-sse/services/contextManager.ts` -- การบังคับใช้ตัวกรอง IP: `open-sse/services/ipFilter.ts` -- การติดตามเซสชัน: `open-sse/services/sessionManager.ts` -- ขอการขจัดข้อมูลซ้ำซ้อน: `open-sse/services/signatureCache.ts` -- การแจ้งระบบ: `open-sse/services/systemPrompt.ts` -- การคิดการจัดการงบประมาณ: `open-sse/services/thinkingBudget.ts` -- การกำหนดเส้นทางโมเดลตัวแทน: `open-sse/services/wildcardRouter.ts` -- การจัดการขีดจำกัดอัตรา: `open-sse/services/rateLimitManager.ts` -- เซอร์กิตเบรกเกอร์: `open-sse/services/circuitBreaker.ts` +Services (business logic): -โมดูลเลเยอร์โดเมน: +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- ความพร้อมใช้งานของโมเดล: `src/lib/domain/modelAvailability.ts` -- กฎต้นทุน/งบประมาณ: `src/lib/domain/costRules.ts` -- นโยบายทางเลือก: `src/lib/domain/fallbackPolicy.ts` -- ตัวแก้ไขคำสั่งผสม: `src/lib/domain/comboResolver.ts` -- นโยบายการล็อก: `src/lib/domain/lockoutPolicy.ts` -- เครื่องมือนโยบาย: `src/domain/policyEngine.ts` — การล็อคแบบรวมศูนย์ → งบประมาณ → การประเมินทางเลือก -- แคตตาล็อกรหัสข้อผิดพลาด: `src/lib/domain/errorCodes.ts` -- รหัสคำขอ: `src/lib/domain/requestId.ts` -- หมดเวลาการดึงข้อมูล: `src/lib/domain/fetchTimeout.ts` -- ขอการตรวจวัดทางไกล: `src/lib/domain/requestTelemetry.ts` -- การปฏิบัติตามข้อกำหนด/การตรวจสอบ: `src/lib/domain/compliance/index.ts` -- นักวิ่ง Eval: `src/lib/domain/evalRunner.ts` -- การคงอยู่ของสถานะโดเมน: `src/lib/db/domainState.ts` — SQLite CRUD สำหรับเชนสำรอง งบประมาณ ประวัติต้นทุน สถานะการล็อกเอาต์ เซอร์กิตเบรกเกอร์ +Domain layer modules: -โมดูลผู้ให้บริการ OAuth (12 ไฟล์แต่ละไฟล์ภายใต้ `src/lib/oauth/providers/`): +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -- ดัชนีรีจิสทรี: `src/lib/oauth/providers/index.ts` -- ผู้ให้บริการส่วนบุคคล: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` -- Thin wrapper: `src/lib/oauth/providers.ts` — ส่งออกซ้ำจากแต่ละโมดูล## 3) Persistence Layer +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -ฐานข้อมูลสถานะหลัก (SQLite): +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -- Core infra: `src/lib/db/core.ts` (ดีกว่า sqlite3, การโยกย้าย, WAL) -- ส่งออกส่วนหน้าอีกครั้ง: `src/lib/localDb.ts` (เลเยอร์ความเข้ากันได้แบบบางสำหรับผู้โทร) -- ไฟล์: `${DATA_DIR}/storage.sqlite` (หรือ `$XDG_CONFIG_HOME/omniroute/storage.sqlite` เมื่อตั้งค่า มิฉะนั้น `~/.omniroute/storage.sqlite`) -- เอนทิตี (ตาราง + เนมสเปซ KV): providerConnections, providerNodes, modelAliases, คอมโบ, apiKeys, การตั้งค่า, การกำหนดราคา,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +## 3) Persistence Layer -ความคงทนในการใช้งาน: +Primary state DB (SQLite): -- ด้านหน้า: `src/lib/usageDb.ts` (แยกส่วนโมดูลใน `src/lib/usage/*`) -- ตาราง SQLite ใน `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` -- สิ่งประดิษฐ์ของไฟล์ทางเลือกยังคงอยู่สำหรับความเข้ากันได้/การแก้ไขข้อบกพร่อง (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- ไฟล์ JSON เดิมจะถูกย้ายไปยัง SQLite โดยการโยกย้ายเริ่มต้นเมื่อมีอยู่ +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -ฐานข้อมูลสถานะโดเมน (SQLite): +Usage persistence: -- `src/lib/db/domainState.ts` — การดำเนินการ CRUD สำหรับสถานะโดเมน -- ตาราง (สร้างใน `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- รูปแบบแคชการเขียนผ่าน: แผนที่ในหน่วยความจำเชื่อถือได้ ณ รันไทม์ การกลายพันธุ์จะถูกเขียนพร้อมกันกับ SQLite; สถานะถูกกู้คืนจาก DB เมื่อสตาร์ทขณะเย็น## 4) Auth + Security Surfaces +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- การตรวจสอบคุกกี้แดชบอร์ด: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- การสร้าง/การตรวจสอบคีย์ API: `src/shared/utils/apiKey.ts` -- ข้อมูลลับของผู้ให้บริการยังคงอยู่ในรายการ `providerConnections` -- รองรับพร็อกซีขาออกผ่าน `open-sse/utils/proxyFetch.ts` (env vars) และ `open-sse/utils/networkProxy.ts` (กำหนดค่าได้ต่อผู้ให้บริการหรือทั่วโลก)## 5) Cloud Sync +Domain State DB (SQLite): -- เริ่มต้นตัวกำหนดเวลา: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- งานประจำ: `src/shared/services/cloudSyncScheduler.ts` -- งานประจำ: `src/shared/services/modelSyncScheduler.ts` -- เส้นทางควบคุม: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -การตัดสินใจทางเลือกนั้นขับเคลื่อนโดย `open-sse/services/accountFallback.ts` โดยใช้รหัสสถานะและการวิเคราะห์พฤติกรรมข้อความแสดงข้อผิดพลาด การกำหนดเส้นทางแบบคอมโบเพิ่มการป้องกันพิเศษหนึ่งรายการ: 400s ที่กำหนดขอบเขตโดยผู้ให้บริการ เช่น ความล้มเหลวในการบล็อกเนื้อหาอัปสตรีมและการตรวจสอบบทบาทจะถือเป็นความล้มเหลวในแบบจำลองภายใน ดังนั้นเป้าหมายคอมโบในภายหลังยังคงสามารถทำงานได้## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -การรีเฟรชระหว่างการรับส่งข้อมูลสดจะดำเนินการภายใน `open-sse/handlers/chatCore.ts` ผ่านทางตัวดำเนินการ `refreshCredentials()`## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -การซิงค์เป็นระยะจะถูกทริกเกอร์โดย `CloudSyncScheduler` เมื่อเปิดใช้งานระบบคลาวด์## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -ไฟล์จัดเก็บข้อมูลทางกายภาพ: +Physical storage files: -- ฐานข้อมูลรันไทม์หลัก: `${DATA_DIR}/storage.sqlite` -- ขอบรรทัดบันทึก: `${DATA_DIR}/log.txt` (compat/debug artifact) -- ไฟล์เก็บถาวรเพย์โหลดการโทรที่มีโครงสร้าง: `${DATA_DIR}/call_logs/` -- ตัวเลือกสำหรับนักแปล/คำขอเซสชันการแก้ไขข้อบกพร่อง: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: API ความเข้ากันได้ -- `src/app/api/v1/providers/[provider]/*`: เส้นทางเฉพาะต่อผู้ให้บริการ (แชท การฝัง รูปภาพ) -- `src/app/api/providers*`: ผู้ให้บริการ CRUD, การตรวจสอบความถูกต้อง, การทดสอบ -- `src/app/api/provider-nodes*`: การจัดการโหนดที่เข้ากันได้แบบกำหนดเอง -- `src/app/api/provider-models`: การจัดการโมเดลแบบกำหนดเอง (CRUD) -- `src/app/api/models/route.ts`: API แคตตาล็อกโมเดล (นามแฝง + โมเดลที่กำหนดเอง) -- `src/app/api/oauth/*`: การไหลของ OAuth/รหัสอุปกรณ์ -- `src/app/api/keys*`: วงจรการใช้งานคีย์ API ภายในเครื่อง -- `src/app/api/models/alias`: การจัดการนามแฝง -- `src/app/api/combos*`: การจัดการคำสั่งผสมสำรอง -- `src/app/api/pricing`: แทนที่การกำหนดราคาสำหรับการคำนวณต้นทุน -- `src/app/api/settings/proxy`: การกำหนดค่าพร็อกซี (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: การทดสอบการเชื่อมต่อพร็อกซีขาออก (POST) -- `src/app/api/usage/*`: การใช้งานและบันทึก API -- `src/app/api/sync/*` + `src/app/api/cloud/*`: การซิงค์บนคลาวด์และผู้ช่วยเหลือบนคลาวด์ -- `src/app/api/cli-tools/*`: ตัวเขียน/ตัวตรวจสอบการกำหนดค่า CLI ในเครื่อง -- `src/app/api/settings/ip-filter`: รายการ IP ที่อนุญาต/รายการบล็อก (GET/PUT) -- `src/app/api/settings/thinking-budget`: กำหนดค่างบประมาณโทเค็นการคิด (GET/PUT) -- `src/app/api/settings/system-prompt`: แจ้งระบบทั่วโลก (GET/PUT) -- `src/app/api/sessions`: รายการเซสชันที่ใช้งานอยู่ (GET) -- `src/app/api/rate-limits`: สถานะขีดจำกัดอัตราต่อบัญชี (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: คำขอแยกวิเคราะห์, การจัดการคำสั่งผสม, วนรอบการเลือกบัญชี -- `open-sse/handlers/chatCore.ts`: การแปล, การส่งตัวดำเนินการ, การจัดการลองใหม่/รีเฟรช, การตั้งค่าสตรีม -- `open-sse/executors/*`: เครือข่ายเฉพาะผู้ให้บริการและพฤติกรรมรูปแบบ### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: รีจิสทรีของนักแปลและการจัดเรียบเรียง -- ขอนักแปล: `open-sse/translator/request/*` -- ผู้แปลคำตอบ: `open-sse/translator/response/*` -- จัดรูปแบบค่าคงที่: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: การกำหนดค่า/สถานะและการคงอยู่ของโดเมนแบบถาวรบน SQLite -- `src/lib/localDb.ts`: ส่งออกความเข้ากันได้อีกครั้งสำหรับโมดูล DB -- `src/lib/usageDb.ts`: ประวัติการใช้งาน/บันทึกการโทรที่อยู่ด้านบนของตาราง SQLite## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -ผู้ให้บริการแต่ละรายมีตัวดำเนินการเฉพาะที่ขยาย `BaseExecutor` (ใน `open-sse/executors/base.ts`) ซึ่งจัดให้มีการสร้าง URL, การสร้างส่วนหัว, ลองอีกครั้งโดยใช้ Exponential Backoff, Hook การรีเฟรชข้อมูลประจำตัว และวิธีการประสาน `execute()` +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| ผู้ดำเนินการ | ผู้ให้บริการ | การจัดการพิเศษ | -| ----------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------- | -| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, ความฉงนสนเท่ห์, Together, ดอกไม้ไฟ, Cerebras, Cohere, NVIDIA | URL แบบไดนามิก/การกำหนดค่าส่วนหัวต่อผู้ให้บริการ | -| `ผู้ดำเนินการต้านแรงโน้มถ่วง` | Google ต้านแรงโน้มถ่วง | รหัสโปรเจ็กต์/เซสชันแบบกำหนดเอง ลองอีกครั้งหลังจากแยกวิเคราะห์ | -| `CodexExecutor` | OpenAI Codex | แทรกคำสั่งของระบบ บังคับใช้ความพยายามในการให้เหตุผล | -| `เคอร์เซอร์เอ็กเซ็กเตอร์` | เคอร์เซอร์ IDE | โปรโตคอล ConnectRPC, การเข้ารหัส Protobuf, ขอการลงนามผ่านเช็คซัม | -| `GithubExecutor` | นักบิน GitHub | การรีเฟรชโทเค็น Copilot ส่วนหัวการเลียนแบบ VSCode | -| `KiroExecutor` | AWS CodeWhisperer/Kiro | รูปแบบไบนารี AWS EventStream → การแปลง SSE | -| `GeminiCLIExecutor` | ราศีเมถุน CLI | วงจรการรีเฟรชโทเค็น Google OAuth | +### Persistence -ผู้ให้บริการรายอื่นทั้งหมด (รวมถึงโหนดที่เข้ากันได้แบบกำหนดเอง) ใช้ `DefaultExecutor`## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| ผู้ให้บริการ | รูปแบบ | รับรองความถูกต้อง | สตรีม | ไม่ใช่สตรีม | รีเฟรชโทเค็น | API การใช้งาน | -| ------------------- | --------------- | ---------------------- | ---------------- | ----------- | ------------ | --------------------- | ------------------------------ | -| คลอดด์ | คลอด | คีย์ API / OAuth | ✅ | ✅ | ✅ | ⚠️เฉพาะแอดมินเท่านั้น | -| ราศีเมถุน | ราศีเมถุน | คีย์ API / OAuth | ✅ | ✅ | ✅ | ⚠️ คลาวด์คอนโซล | -| ราศีเมถุน CLI | ราศีเมถุน-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ คลาวด์คอนโซล | -| ต้านแรงโน้มถ่วง | ต้านแรงโน้มถ่วง | OAuth | ✅ | ✅ | ✅ | ✅ API โควต้าเต็ม | -| OpenAI | เปิดใจ | คีย์ API | ✅ | ✅ | ❌ | ❌ | -| โคเด็กซ์ | openai ตอบกลับ | OAuth | ✅บังคับ | ❌ | ✅ | ✅ ขีดจำกัดอัตรา | -| นักบิน GitHub | เปิดใจ | OAuth + โทเค็น Copilot | ✅ | ✅ | ✅ | ✅ สแนปชอตโควต้า | -| เคอร์เซอร์ | เคอร์เซอร์ | เช็คซัมแบบกำหนดเอง | ✅ | ✅ | ❌ | ❌ | -| คิโระ | คิโระ | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ ขีดจำกัดการใช้งาน | -| ควีน | เปิดใจ | OAuth | ✅ | ✅ | ✅ | ⚠️ตามคำขอ | -| คิวเดอร์ | เปิดใจ | OAuth (พื้นฐาน) | ✅ | ✅ | ✅ | ⚠️ตามคำขอ | -| OpenRouter | เปิดใจ | คีย์ API | ✅ | ✅ | ❌ | ❌ | -| GLM/คิมิ/มินิแม็กซ์ | คลอด | คีย์ API | ✅ | ✅ | ❌ | ❌ | -| DeepSeek | เปิดใจ | คีย์ API | ✅ | ✅ | ❌ | ❌ | -| กรอค | เปิดใจ | คีย์ API | ✅ | ✅ | ❌ | ❌ | -| xAI (โกรก) | เปิดใจ | คีย์ API | ✅ | ✅ | ❌ | ❌ | -| มิสทรัล | เปิดใจ | คีย์ API | ✅ | ✅ | ❌ | ❌ | -| ความฉงนสนเท่ห์ | เปิดใจ | คีย์ API | ✅ | ✅ | ❌ | ❌ | -| ร่วมกัน AI | เปิดใจ | คีย์ API | ✅ | ✅ | ❌ | ❌ | -| ดอกไม้ไฟ AI | เปิดใจ | คีย์ API | ✅ | ✅ | ❌ | ❌ | -| สมอง | เปิดใจ | คีย์ API | ✅ | ✅ | ❌ | ❌ | -| เชื่อมโยง | เปิดใจ | คีย์ API | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | เปิดใจ | คีย์ API | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -รูปแบบแหล่งที่มาที่ตรวจพบ ได้แก่: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. -- `เปิดใจ` -- `การตอบกลับแบบ openai' -- 'โคลด' -- `ราศีเมถุน` +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | -รูปแบบเป้าหมายได้แก่: +All other providers (including custom compatible nodes) use the `DefaultExecutor`. -- แชท / ตอบกลับ OpenAI -- คลอดด์ -- Gemini/Gemini-CLI/ซองต้านแรงโน้มถ่วง -- คิโระ -- เคอร์เซอร์ +## Provider Compatibility Matrix -การแปลใช้**OpenAI เป็นรูปแบบฮับ**— การแปลงทั้งหมดผ่าน OpenAI เป็นตัวกลาง:``` +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: + +- `openai` +- `openai-responses` +- `claude` +- `gemini` + +Target formats include: + +- OpenAI chat/Responses +- Claude +- Gemini/Gemini-CLI/Antigravity envelope +- Kiro +- Cursor + +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -การแปลจะถูกเลือกแบบไดนามิกตามรูปร่างเพย์โหลดต้นทางและรูปแบบเป้าหมายของผู้ให้บริการ +Additional processing layers in the translation pipeline: -เลเยอร์การประมวลผลเพิ่มเติมในไปป์ไลน์การแปล: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**การฆ่าเชื้อการตอบสนอง**— ตัดช่องที่ไม่ได้มาตรฐานออกจากการตอบสนองในรูปแบบ OpenAI (ทั้งแบบสตรีมมิ่งและไม่ใช่สตรีมมิ่ง) เพื่อให้มั่นใจว่าสอดคล้องกับ SDK ที่เข้มงวด --**การปรับบทบาทให้เป็นมาตรฐาน**— แปลง `ผู้พัฒนา` → `ระบบ` สำหรับเป้าหมายที่ไม่ใช่ OpenAI ผสาน `system` → `user` สำหรับโมเดลที่ปฏิเสธบทบาทของระบบ (GLM, ERNIE) --**คิดว่าการแยกแท็ก**— แยกวิเคราะห์บล็อก `...` จากเนื้อหาลงในช่อง `reasoning_content` --**เอาต์พุตที่มีโครงสร้าง**— แปลง OpenAI `response_format.json_schema` เป็น `responseMimeType` + `responseSchema` ของ Gemini## Supported API Endpoints +## Supported API Endpoints -| จุดสิ้นสุด | รูปแบบ | ตัวจัดการ | +| Endpoint | Format | Handler | | -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | -| `โพสต์ /v1/แชท/เสร็จสิ้น` | แชท OpenAI | `src/sse/handlers/chat.ts` | -| `โพสต์ /v1/ข้อความ` | ข้อความของคลอดด์ | ตัวจัดการเดียวกัน (ตรวจพบอัตโนมัติ) | -| `POST /v1/ตอบกลับ` | การตอบสนองของ OpenAI | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/การฝัง` | การฝัง OpenAI | `open-sse/handlers/embeddings.ts` | -| `GET /v1/การฝัง` | รายการรุ่น | เส้นทาง API | -| `POST /v1/images/รุ่น` | รูปภาพ OpenAI | `open-sse/handlers/imageGeneration.ts` | -| `รับ /v1/รูปภาพ/รุ่น` | รายการรุ่น | เส้นทาง API | -| `POST /v1/providers/{provider}/chat/completions` | แชท OpenAI | เฉพาะต่อผู้ให้บริการพร้อมการตรวจสอบโมเดล | -| `POST /v1/providers/{provider}/embeddings` | การฝัง OpenAI | เฉพาะต่อผู้ให้บริการพร้อมการตรวจสอบโมเดล | -| `POST /v1/providers/{provider}/images/รุ่น` | รูปภาพ OpenAI | เฉพาะต่อผู้ให้บริการพร้อมการตรวจสอบโมเดล | -| `POST /v1/messages/count_tokens` | จำนวนโทเค็นของ Claude | เส้นทาง API | -| `GET /v1/models` | รายการโมเดล OpenAI | เส้นทาง API (แชท + การฝัง + รูปภาพ + โมเดลที่กำหนดเอง) | -| `GET /api/models/catalog` | แคตตาล็อก | ทุกรุ่นจัดกลุ่มตามผู้ให้บริการ + ประเภท | -| `POST /v1beta/models/*:streamGenerateContent` | ชาวราศีเมถุนพื้นเมือง | เส้นทาง API | -| `รับ/วาง/ลบ /api/การตั้งค่า/พร็อกซี` | การกำหนดค่าพร็อกซี | การกำหนดค่าพร็อกซีเครือข่าย | -| `POST /api/settings/proxy/test` | การเชื่อมต่อพร็อกซี | จุดสิ้นสุดการทดสอบความสมบูรณ์ของพร็อกซี/การเชื่อมต่อ | -| `GET/POST/DELETE /api/provider-models` | โมเดลผู้ให้บริการ | ผู้ให้บริการสนับสนุนข้อมูลเมตาโมเดลที่กำหนดเองและโมเดลที่มีการจัดการ |## Bypass Handler +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -ตัวจัดการบายพาส (`open-sse/utils/bypassHandler.ts`) สกัดกั้นคำขอ "ที่ใช้แล้วทิ้ง" ที่รู้จักจาก Claude CLI เช่น การปิงการอุ่นเครื่อง การแตกชื่อ และจำนวนโทเค็น และส่งคืน**การตอบกลับปลอม**โดยไม่ต้องใช้โทเค็นของผู้ให้บริการอัปสตรีม ซึ่งจะเกิดขึ้นเมื่อ `User-Agent` มี `claude-cli` เท่านั้น## Request Logger Pipeline +## Bypass Handler -ตัวบันทึกคำขอ (`open-sse/utils/requestLogger.ts`) จัดเตรียมไปป์ไลน์การบันทึกการดีบัก 7 ขั้นตอน ซึ่งปิดใช้งานตามค่าเริ่มต้น เปิดใช้งานผ่าน `ENABLE_REQUEST_LOGS=true`:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -ไฟล์ถูกเขียนไปที่ `/logs//` สำหรับแต่ละเซสชันคำขอ## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- คูลดาวน์บัญชีผู้ให้บริการเกี่ยวกับข้อผิดพลาดชั่วคราว/อัตรา/การตรวจสอบสิทธิ์ -- ทางเลือกบัญชีก่อนที่จะล้มเหลวในการร้องขอ -- ทางเลือกของโมเดลคอมโบเมื่อพาธของโมเดล/ผู้ให้บริการปัจจุบันหมดลง## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- ตรวจสอบล่วงหน้าและรีเฟรชด้วยการลองอีกครั้งสำหรับผู้ให้บริการที่รีเฟรชได้ -- 401/403 ลองอีกครั้งหลังจากพยายามรีเฟรชในเส้นทางหลัก## 3) Stream Safety +## 2) Token Expiry -- ตัวควบคุมสตรีมที่รับรู้การตัดการเชื่อมต่อ -- สตรีมการแปลพร้อมฟลัชปลายสตรีมและการจัดการ `[เสร็จสิ้น]` -- ทางเลือกการประมาณการใช้งานเมื่อข้อมูลเมตาการใช้งานของผู้ให้บริการหายไป## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- ข้อผิดพลาดในการซิงค์ปรากฏขึ้น แต่รันไทม์ในเครื่องยังคงดำเนินต่อไป -- ตัวกำหนดตารางเวลามีตรรกะที่สามารถลองใหม่ได้ แต่การดำเนินการตามระยะเวลาในปัจจุบันจะเรียกการซิงค์แบบพยายามครั้งเดียวตามค่าเริ่มต้น## 5) Data Integrity +## 3) Stream Safety -- การโยกย้ายสคีมา SQLite และ hooks อัปเกรดอัตโนมัติเมื่อเริ่มต้น -- JSON ดั้งเดิม → เส้นทางความเข้ากันได้ของการโยกย้าย SQLite## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -แหล่งที่มาของการมองเห็นรันไทม์: +## 4) Cloud Sync Degradation -- บันทึกคอนโซลจาก `src/sse/utils/logger.ts` -- การรวมการใช้งานต่อคำขอใน SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- การจับเพย์โหลดโดยละเอียดสี่ขั้นตอนใน SQLite (`request_detail_logs`) เมื่อ `settings.detailed_logs_enabled=true` -- บันทึกสถานะคำขอที่เป็นข้อความใน `log.txt` (ตัวเลือก / เข้ากันได้) -- บันทึกคำขอ/การแปลเชิงลึกที่เป็นทางเลือกภายใต้ `logs/` เมื่อ `ENABLE_REQUEST_LOGS=true` -- จุดสิ้นสุดการใช้งานแดชบอร์ด (`/api/usage/*`) สำหรับการใช้ UI +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -การบันทึกเพย์โหลดคำขอโดยละเอียดจะจัดเก็บขั้นตอนเพย์โหลด JSON ได้สูงสุดสี่ขั้นตอนต่อการโทรที่กำหนดเส้นทาง: +## 5) Data Integrity -- คำขอดิบที่ได้รับจากลูกค้า -- คำขอที่แปลแล้วส่งต้นทางจริง -- การตอบสนองของผู้ให้บริการสร้างใหม่เป็น JSON การตอบกลับแบบสตรีมจะถูกบีบอัดเป็นข้อมูลสรุปสุดท้ายพร้อมข้อมูลเมตาของสตรีม -- การตอบสนองของลูกค้าขั้นสุดท้ายที่ส่งคืนโดย OmniRoute; การตอบกลับแบบสตรีมจะถูกจัดเก็บไว้ในแบบฟอร์มสรุปแบบกะทัดรัดเดียวกัน## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- ความลับ JWT (`JWT_SECRET`) รักษาความปลอดภัยการตรวจสอบ / การลงนามคุกกี้เซสชันแดชบอร์ด -- บูตสแตรปรหัสผ่านเริ่มต้น (`INITIAL_PASSWORD`) ควรได้รับการกำหนดค่าอย่างชัดเจนสำหรับการจัดเตรียมการทำงานครั้งแรก -- คีย์ API ความลับ HMAC (`API_KEY_SECRET`) รักษาความปลอดภัยให้กับรูปแบบคีย์ API ในเครื่องที่สร้างขึ้น -- ความลับของผู้ให้บริการ (คีย์/โทเค็น API) ยังคงอยู่ในฐานข้อมูลในเครื่องและควรได้รับการปกป้องในระดับระบบไฟล์ -- จุดสิ้นสุดการซิงค์บนคลาวด์อาศัยการตรวจสอบสิทธิ์คีย์ API + ซีแมนทิกส์รหัสเครื่อง## Environment and Runtime Matrix +## Observability and Operational Signals -ตัวแปรสภาพแวดล้อมที่ใช้งานโดยโค้ด: +Runtime visibility sources: -- แอป/การรับรองความถูกต้อง: `JWT_SECRET`, `INITIAL_PASSWORD` -- พื้นที่เก็บข้อมูล: `DATA_DIR` -- ลักษณะการทำงานของโหนดที่เข้ากันได้: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- การแทนที่ฐานจัดเก็บข้อมูลเสริม (Linux/macOS เมื่อไม่ได้ตั้งค่า `DATA_DIR`): `XDG_CONFIG_HOME` -- การรักษาความปลอดภัย: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- การบันทึก: `ENABLE_REQUEST_LOGS` -- การซิงค์/คลาวด์ URL: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- พร็อกซีขาออก: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` และรูปแบบตัวพิมพ์เล็ก -- ธงคุณลักษณะ SOCKS5: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- ตัวช่วยแพลตฟอร์ม/รันไทม์ (ไม่ใช่การกำหนดค่าเฉพาะแอป): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` และ `localDb` ใช้นโยบายไดเรกทอรีฐานเดียวกัน (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) พร้อมการย้ายไฟล์แบบเดิม -2. `/api/v1/route.ts` มอบสิทธิ์ให้กับตัวสร้างแคตตาล็อกแบบรวมตัวเดียวกันกับที่ใช้โดย `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) เพื่อหลีกเลี่ยงการเบี่ยงเบนทางความหมาย -3. ตัวบันทึกคำขอเขียนส่วนหัว/เนื้อหาแบบเต็มเมื่อเปิดใช้งาน ถือว่าไดเร็กทอรีบันทึกมีความละเอียดอ่อน -4. พฤติกรรมของคลาวด์ขึ้นอยู่กับ `NEXT_PUBLIC_BASE_URL` ที่ถูกต้องและการเข้าถึงจุดสิ้นสุดของคลาวด์ -5. ไดเรกทอรี `open-sse/` ได้รับการเผยแพร่เป็น `@omniroute/open-sse`**แพ็คเกจพื้นที่ทำงาน npm**ซอร์สโค้ดนำเข้าผ่าน `@omniroute/open-sse/...` (แก้ไขโดย Next.js `transpilePackages`) เส้นทางของไฟล์ในเอกสารนี้ยังคงใช้ชื่อไดเรกทอรี `open-sse/` เพื่อความสอดคล้อง -6. แผนภูมิในแดชบอร์ดใช้**แผนภูมิใหม่**(อิงตาม SVG) สำหรับการแสดงภาพการวิเคราะห์เชิงโต้ตอบที่เข้าถึงได้ (แผนภูมิแท่งการใช้งานโมเดล ตารางแจกแจงผู้ให้บริการพร้อมอัตราความสำเร็จ) -7. การทดสอบ E2E ใช้**นักเขียนบทละคร**(`tests/e2e/`) รันผ่าน `npm run test:e2e` การทดสอบหน่วยใช้**ตัวรันการทดสอบ Node.js**(`tests/unit/`) รันผ่าน `npm run test:unit` ซอร์สโค้ดภายใต้ `src/` คือ**TypeScript**(`.ts`/`.tsx`); พื้นที่ทำงาน `open-sse/` ยังคงเป็น JavaScript (`.js`) -8. หน้าการตั้งค่าแบ่งออกเป็น 5 แท็บ: ความปลอดภัย การกำหนดเส้นทาง (6 กลยุทธ์ระดับโลก: เติมก่อน ปัดเศษ p2c สุ่ม ใช้น้อยที่สุด ปรับต้นทุนให้เหมาะสม) ความยืดหยุ่น (จำกัดอัตราที่แก้ไขได้ เซอร์กิตเบรกเกอร์ นโยบาย) AI (การคิดงบประมาณ พรอมต์ของระบบ แคชพร้อมต์) ขั้นสูง (พร็อกซี)## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- สร้างจากแหล่งที่มา: `npm run build` -- สร้างอิมเมจ Docker: `docker build -t omniroute .` -- เริ่มบริการและตรวจสอบ: -- `รับ /api/การตั้งค่า` +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` - `GET /api/v1/models` -- URL ฐานเป้าหมาย CLI ควรเป็น `http://:20128/v1` เมื่อ `PORT=20128` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/th/docs/FEATURES.md b/docs/i18n/th/docs/FEATURES.md index 73571a4a6b..36b50767f3 100644 --- a/docs/i18n/th/docs/FEATURES.md +++ b/docs/i18n/th/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -ภาพแนะนำทุกส่วนของแดชบอร์ด OmniRoute--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -จัดการการเชื่อมต่อผู้ให้บริการ AI: ผู้ให้บริการ OAuth (Claude Code, Codex, Gemini CLI), ผู้ให้บริการคีย์ API (Groq, DeepSeek, OpenRouter) และผู้ให้บริการฟรี (Qoder, Qwen, Kiro) บัญชี Kiro มีการติดตามยอดเครดิต — เครดิตที่เหลือ เบี้ยเลี้ยงทั้งหมด และวันที่ต่ออายุสามารถดูได้ในแดชบอร์ด → การใช้งาน![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -สร้างคอมโบการกำหนดเส้นทางแบบจำลองด้วย 6 กลยุทธ์: ลำดับความสำคัญ, ถ่วงน้ำหนัก, วนรอบ, สุ่ม, ใช้น้อยที่สุด และปรับต้นทุนให้เหมาะสม แต่ละคอมโบเชื่อมโยงหลายรุ่นด้วยทางเลือกอัตโนมัติ และรวมถึงเทมเพลตที่รวดเร็วและการตรวจสอบความพร้อม![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -การวิเคราะห์การใช้งานที่ครอบคลุมด้วยการใช้โทเค็น การประมาณการต้นทุน แผนที่ความร้อนของกิจกรรม แผนภูมิการกระจายรายสัปดาห์ และรายละเอียดต่อผู้ให้บริการ![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -การตรวจสอบแบบเรียลไทม์: เวลาทำงาน หน่วยความจำ เวอร์ชัน เปอร์เซ็นต์ไทล์แฝง (p50/p95/p99) สถิติแคช และสถานะเซอร์กิตเบรกเกอร์ของผู้ให้บริการ![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -สี่โหมดสำหรับการดีบักการแปล API:**Playground**(ตัวแปลงรูปแบบ),**Chat Tester**(คำขอสด),**Test Bench**(การทดสอบเป็นกลุ่ม) และ**Live Monitor**(สตรีมแบบเรียลไทม์)![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -ทดสอบรุ่นใดก็ได้โดยตรงจากแดชบอร์ด เลือกผู้ให้บริการ โมเดล และจุดสิ้นสุด เขียนพร้อมท์ด้วย Monaco Editor สตรีมการตอบกลับแบบเรียลไทม์ ยกเลิกระหว่างสตรีม และดูตัวชี้วัดเวลา--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -ธีมสีที่ปรับแต่งได้สำหรับแดชบอร์ดทั้งหมด เลือกจากสีที่ตั้งไว้ล่วงหน้า 7 สี (คอรัล, น้ำเงิน, แดง, เขียว, ม่วง, ส้ม, ฟ้า) หรือสร้างธีมแบบกำหนดเองโดยเลือกสีฐานสิบหกใดก็ได้ รองรับโหมดแสง มืด และระบบ--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -แผงการตั้งค่าที่ครอบคลุมพร้อมแท็บ: +Comprehensive settings panel with tabs: --**ทั่วไป**— ที่เก็บข้อมูลระบบ การจัดการการสำรองข้อมูล (ฐานข้อมูลส่งออก/นำเข้า) -**รูปลักษณ์**— ตัวเลือกธีม (มืด/สว่าง/ระบบ) การตั้งค่าธีมสีและสีที่กำหนดเอง การมองเห็นบันทึกสุขภาพ การควบคุมการมองเห็นรายการในแถบด้านข้าง -**ความปลอดภัย**— การป้องกันจุดสิ้นสุด API, การบล็อกผู้ให้บริการแบบกำหนดเอง, การกรอง IP, ข้อมูลเซสชัน -**การกำหนดเส้นทาง**— นามแฝงโมเดล การลดระดับงานในเบื้องหลัง -**ความยืดหยุ่น**— การคงอยู่ของอัตราจำกัด การปรับเซอร์กิตเบรกเกอร์ บัญชีที่ถูกแบนปิดการใช้งานอัตโนมัติ การตรวจสอบการหมดอายุของผู้ให้บริการ -**ขั้นสูง**— การแทนที่การกำหนดค่า เส้นทางการตรวจสอบการกำหนดค่า โหมดการลดประสิทธิภาพทางเลือก![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -การกำหนดค่าเพียงคลิกเดียวสำหรับเครื่องมือเข้ารหัส AI: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor และ Factory Droid นำเสนอการตั้งค่า/รีเซ็ตอัตโนมัติ โปรไฟล์การเชื่อมต่อ และการแมปโมเดล![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -แดชบอร์ดสำหรับการค้นหาและจัดการตัวแทน CLI แสดงตารางของเอเจนต์ในตัว 14 รายการ (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) ด้วย: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**สถานะการติดตั้ง**— ติดตั้งแล้ว / ไม่พบด้วยการตรวจหาเวอร์ชัน -**ป้ายโปรโตคอล**— stdio, HTTP ฯลฯ -**ตัวแทนที่กำหนดเอง**— ลงทะเบียนเครื่องมือ CLI ใด ๆ ผ่านแบบฟอร์ม (ชื่อ, ไบนารี่, คำสั่งเวอร์ชัน, spawn args) -**การจับคู่ลายนิ้วมือ CLI**— สลับระหว่างผู้ให้บริการแต่ละรายเพื่อให้ตรงกับลายเซ็นคำขอ CLI ดั้งเดิม ซึ่งช่วยลดความเสี่ยงในการแบนในขณะที่รักษา IP พร็อกซีไว้--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -สร้างภาพ วิดีโอ และเพลงจากแดชบอร์ด รองรับ OpenAI, xAI, Together, ไฮเปอร์โบลิก, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open และ MusicGen--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -การบันทึกคำขอแบบเรียลไทม์พร้อมการกรองตามผู้ให้บริการ โมเดล บัญชี และคีย์ API แสดงรหัสสถานะ การใช้โทเค็น เวลาแฝง และรายละเอียดการตอบกลับ![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -ตำแหน่งข้อมูล API แบบรวมของคุณพร้อมรายละเอียดความสามารถ: การแชทให้เสร็จสิ้น, API การตอบกลับ, การฝัง, การสร้างรูปภาพ, การจัดอันดับใหม่, การถอดเสียง, การอ่านออกเสียงข้อความ, การกลั่นกรอง และคีย์ API ที่ลงทะเบียน การรวม Cloudflare Quick Tunnel และการสนับสนุนพร็อกซีคลาวด์สำหรับการเข้าถึงระยะไกล![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -สร้าง กำหนดขอบเขต และเพิกถอนคีย์ API แต่ละคีย์สามารถจำกัดเฉพาะรุ่น/ผู้ให้บริการเฉพาะที่มีสิทธิ์การเข้าถึงแบบเต็มหรือสิทธิ์อ่านอย่างเดียว การจัดการคีย์ภาพพร้อมการติดตามการใช้งาน--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -การติดตามการดำเนินการด้านการดูแลระบบพร้อมการกรองตามประเภทการดำเนินการ ผู้ดำเนินการ เป้าหมาย ที่อยู่ IP และการประทับเวลา ประวัติเหตุการณ์ความปลอดภัยแบบเต็ม--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -แอพเดสก์ท็อป Native Electron สำหรับ Windows, macOS และ Linux เรียกใช้ OmniRoute เป็นแอปพลิเคชันแบบสแตนด์อโลนที่มีการบูรณาการถาดระบบ การสนับสนุนแบบออฟไลน์ การอัปเดตอัตโนมัติ และการติดตั้งในคลิกเดียว +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -คุณสมบัติที่สำคัญ: +Key features: -- การโพลความพร้อมของเซิร์ฟเวอร์ (ไม่มีหน้าจอว่างเมื่อสตาร์ทเย็น) -- ถาดระบบพร้อมการจัดการพอร์ต -- นโยบายการรักษาความปลอดภัยของเนื้อหา -- ล็อคอินสแตนซ์เดียว -- อัปเดตอัตโนมัติเมื่อรีสตาร์ท -- UI แบบมีเงื่อนไขของแพลตฟอร์ม (ไฟจราจร macOS, แถบหัวเรื่องเริ่มต้นของ Windows/Linux) -- บรรจุภัณฑ์สำหรับการสร้างอิเล็กตรอนที่แข็งตัว — `node_modules` ที่เชื่อมโยงกันในชุดรวมแบบสแตนด์อโลนจะถูกตรวจจับและปฏิเสธก่อนการบรรจุ ป้องกันการพึ่งพารันไทม์กับเครื่องประกอบ (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 ดู [`electron/README.md`](../electron/README.md) สำหรับเอกสารฉบับเต็ม +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/th/docs/TROUBLESHOOTING.md b/docs/i18n/th/docs/TROUBLESHOOTING.md index 0cfe65cd2b..625fd64113 100644 --- a/docs/i18n/th/docs/TROUBLESHOOTING.md +++ b/docs/i18n/th/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -ปัญหาและวิธีแก้ปัญหาทั่วไปสำหรับ OmniRoute--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| ปัญหา | โซลูชั่น | -| ------------------------------- | ---------------------------------------------------------------------- | --- | -| การเข้าสู่ระบบครั้งแรกไม่ทำงาน | ตั้งค่า `INITIAL_PASSWORD` ใน `.env` (ไม่มีค่าเริ่มต้นแบบฮาร์ดโค้ด) | -| แดชบอร์ดเปิดบนพอร์ตผิด | ตั้งค่า `PORT=20128` และ `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| ไม่มีบันทึกคำขอภายใต้ `logs/` | ตั้งค่า `ENABLE_REQUEST_LOGS=true` | -| EACCES: การอนุญาตถูกปฏิเสธ | ตั้งค่า `DATA_DIR=/path/to/writable/dir` เพื่อแทนที่ `~/.omniroute` | -| กลยุทธ์การกำหนดเส้นทางไม่บันทึก | อัปเดตเป็น v1.4.11+ (แก้ไข Zod schema สำหรับการคงอยู่ของการตั้งค่า) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**สาเหตุ:**โควต้าของผู้ให้บริการหมดลง +**Cause:** Provider quota exhausted. -**แก้ไข:** +**Fix:** -1. ตรวจสอบตัวติดตามโควต้าแดชบอร์ด -2. ใช้คอมโบที่มีระดับทางเลือก -3. เปลี่ยนไปใช้ระดับที่ถูกกว่า/ฟรี### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**สาเหตุ:**โควต้าการสมัครใช้งานหมดลง +### Rate Limiting -**แก้ไข:** +**Cause:** Subscription quota exhausted. -- เพิ่มทางเลือก: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- ใช้ GLM/MiniMax เป็นข้อมูลสำรองราคาถูก### OAuth Token Expired +**Fix:** -OmniRoute รีเฟรชโทเค็นอัตโนมัติ หากปัญหายังคงมีอยู่: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. แดชบอร์ด → ผู้ให้บริการ → เชื่อมต่อใหม่ -2. ลบและเพิ่มการเชื่อมต่อผู้ให้บริการอีกครั้ง--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. ตรวจสอบ `BASE_URL` ชี้ไปยังอินสแตนซ์ที่ทำงานอยู่ของคุณ (เช่น `http://localhost:20128`) -2. ตรวจสอบ `CLOUD_URL` ชี้ไปยังจุดสิ้นสุดระบบคลาวด์ของคุณ (เช่น `https://omniroute.dev`) -3. รักษาค่า `NEXT_PUBLIC_*` ให้สอดคล้องกับค่าฝั่งเซิร์ฟเวอร์### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**อาการ:**`โทเค็นที่ไม่คาดคิด 'd'...` บนจุดปลายทางคลาวด์สำหรับการโทรที่ไม่ใช่การสตรีม +### Cloud `stream=false` Returns 500 -**สาเหตุ:**อัปสตรีมส่งคืนเพย์โหลด SSE ในขณะที่ไคลเอ็นต์คาดหวัง JSON +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**วิธีแก้ปัญหา:**ใช้ `stream=true` สำหรับการโทรโดยตรงบนคลาวด์ รันไทม์ในเครื่องรวมถึงทางเลือก SSE → JSON### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. สร้างคีย์ใหม่จากแดชบอร์ดในเครื่อง (`/api/keys`) -2. เรียกใช้คลาวด์ซิงค์: เปิดใช้งานคลาวด์ → ซิงค์ทันที -3. คีย์เก่า/ที่ไม่ได้ซิงค์ยังสามารถส่งคืน `401` บนคลาวด์ได้--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. ตรวจสอบฟิลด์รันไทม์: `curl http://localhost:20128/api/cli-tools/runtime/codex | เจคิว` -2. สำหรับโหมดพกพา: ใช้เป้าหมายรูปภาพ `runner-cli` (CLI ที่รวมมาด้วย) -3. สำหรับโหมดเมานท์โฮสต์: ตั้งค่า `CLI_EXTRA_PATHS` และเมานต์ไดเร็กทอรี bin โฮสต์เป็นแบบอ่านอย่างเดียว -4. หาก `installed=true` และ `runnable=false`: พบไบนารีแต่ตรวจสุขภาพไม่สำเร็จ### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. ตรวจสอบสถิติการใช้งานในแดชบอร์ด → การใช้งาน -2. สลับโมเดลหลักเป็น GLM/MiniMax -3. ใช้ Free Tier (Gemini CLI, Qoder) สำหรับงานที่ไม่สำคัญ -4. กำหนดงบประมาณต้นทุนต่อคีย์ API: แดชบอร์ด → คีย์ API → งบประมาณ--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -ตั้งค่า `ENABLE_REQUEST_LOGS=true` ในไฟล์ `.env` ของคุณ บันทึกจะปรากฏใต้ไดเรกทอรี `logs/`### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- สถานะหลัก: `${DATA_DIR}/storage.sqlite` (ผู้ให้บริการ คอมโบ นามแฝง คีย์ การตั้งค่า) -- การใช้งาน: ตาราง SQLite ใน `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + ตัวเลือก `${DATA_DIR}/log.txt` และ `${DATA_DIR}/call_logs/` -- ขอบันทึก: `/logs/...` (เมื่อ `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -เมื่อเซอร์กิตเบรกเกอร์ของผู้ให้บริการเปิดอยู่ คำขอจะถูกบล็อกจนกว่าคูลดาวน์จะหมดลง +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**แก้ไข:** +**Fix:** -1. ไปที่**แดชบอร์ด → การตั้งค่า → ความยืดหยุ่น** -2. ตรวจสอบการ์ดเซอร์กิตเบรกเกอร์สำหรับผู้ให้บริการที่ได้รับผลกระทบ -3. คลิก**รีเซ็ตทั้งหมด**เพื่อล้างเบรกเกอร์ทั้งหมด หรือรอให้คูลดาวน์หมดลง -4. ตรวจสอบว่าผู้ให้บริการพร้อมใช้งานจริงก่อนที่จะรีเซ็ต### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -หากผู้ให้บริการเข้าสู่สถานะเปิดซ้ำๆ: +### Provider keeps tripping the circuit breaker -1. ตรวจสอบ**แดชบอร์ด → สุขภาพ → สุขภาพของผู้ให้บริการ**เพื่อดูรูปแบบความล้มเหลว -2. ไปที่**การตั้งค่า → ความยืดหยุ่น → โปรไฟล์ผู้ให้บริการ**และเพิ่มเกณฑ์ความล้มเหลว -3. ตรวจสอบว่าผู้ให้บริการได้เปลี่ยนแปลงขีดจำกัด API หรือต้องมีการตรวจสอบสิทธิ์อีกครั้งหรือไม่ -4. ตรวจสอบการวัดและส่งข้อมูลทางไกลเวลาแฝง — เวลาแฝงสูงอาจทำให้เกิดความล้มเหลวตามการหมดเวลา--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- ตรวจสอบให้แน่ใจว่าคุณใช้คำนำหน้าที่ถูกต้อง: `deepgram/nova-3` หรือ `assemblyai/best` -- ตรวจสอบว่าผู้ให้บริการเชื่อมต่ออยู่ใน**Dashboard → Providers**### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- ตรวจสอบรูปแบบเสียงที่รองรับ: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- ตรวจสอบขนาดไฟล์อยู่ภายในขีดจำกัดของผู้ให้บริการ (โดยทั่วไปคือ <25MB) -- ตรวจสอบความถูกต้องของคีย์ API ของผู้ให้บริการในการ์ดผู้ให้บริการ--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -ใช้**แดชบอร์ด → ตัวแปล**เพื่อแก้ไขปัญหาการแปลรูปแบบ: +Use **Dashboard → Translator** to debug format translation issues: -| โหมด | เมื่อใดควรใช้ | -| ------------------------- | ------------------------------------------------------------------------------------ | ------------------------ | -| **สนามเด็กเล่น** | เปรียบเทียบรูปแบบอินพุต/เอาต์พุตแบบเคียงข้างกัน — วางคำขอที่ล้มเหลวเพื่อดูว่าคำขอแปล | อย่างไร | -| **เครื่องมือทดสอบการแชท** | ส่งข้อความสดและตรวจสอบเพย์โหลดคำขอ/การตอบกลับทั้งหมด รวมถึงส่วนหัว | -| **ม้านั่งทดสอบ** | เรียกใช้การทดสอบเป็นชุดระหว่างรูปแบบต่างๆ เพื่อดูว่าคำแปลใดเสียหาย | -| **ถ่ายทอดสด** | ดูขั้นตอนคำขอแบบเรียลไทม์เพื่อตรวจจับปัญหาการแปลเป็นระยะๆ | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**แท็กการคิดไม่ปรากฏ**— ตรวจสอบว่าผู้ให้บริการเป้าหมายสนับสนุนการคิดและการตั้งค่างบประมาณการคิดหรือไม่ -**การเรียกเครื่องมือลดลง**— การแปลรูปแบบบางรูปแบบอาจตัดช่องที่ไม่รองรับออก ตรวจสอบในโหมดสนามเด็กเล่น -**การแจ้งเตือนของระบบหายไป**— ระบบแจ้งของ Claude และ Gemini แตกต่างกัน ตรวจสอบผลลัพธ์การแปล -**SDK ส่งคืนสตริงดิบแทนที่จะเป็นวัตถุ**— แก้ไขใน v1.1.0: ตอนนี้ตัวล้างการตอบสนองจะตัดฟิลด์ที่ไม่ได้มาตรฐาน (`x_groq`, `usage_breakdown` ฯลฯ ) ที่ทำให้การตรวจสอบ OpenAI SDK Pydantic ล้มเหลว -**GLM/ERNIE ปฏิเสธ `บทบาทของระบบ`**- แก้ไขใน v1.1.0: บทบาท Normalizer จะรวมข้อความของระบบเข้ากับข้อความผู้ใช้โดยอัตโนมัติสำหรับรุ่นที่เข้ากันไม่ได้ -**``ไม่รู้จักบทบาทของนักพัฒนา'' - แก้ไขใน v1.1.0: แปลงเป็น `ระบบ` โดยอัตโนมัติสำหรับผู้ให้บริการที่ไม่ใช่ OpenAI -**`json_schema` ไม่ทำงานกับ Gemini\*\*- แก้ไขใน v1.1.0: ตอนนี้ `response_format` ถูกแปลงเป็น `responseMimeType` ของ Gemini + `responseSchema`--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- การจำกัดอัตราอัตโนมัติใช้กับผู้ให้บริการคีย์ API เท่านั้น (ไม่ใช่ OAuth/การสมัครสมาชิก) -- ตรวจสอบ**การตั้งค่า → ความยืดหยุ่น → โปรไฟล์ผู้ให้บริการ**ได้เปิดใช้งานการจำกัดอัตราอัตโนมัติแล้ว -- ตรวจสอบว่าผู้ให้บริการส่งคืนรหัสสถานะ '429` หรือส่วนหัว 'Retry-After'### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -โปรไฟล์ผู้ให้บริการรองรับการตั้งค่าเหล่านี้: +### Tuning exponential backoff --**ความล่าช้าพื้นฐาน**— เวลารอเริ่มต้นหลังจากความล้มเหลวครั้งแรก (ค่าเริ่มต้น: 1 วินาที) -**ความล่าช้าสูงสุด**— ขีดจำกัดเวลารอสูงสุด (ค่าเริ่มต้น: 30 วินาที) -**ตัวคูณ**— จะต้องเพิ่มความล่าช้าเท่าใดต่อความล้มเหลวติดต่อกัน (ค่าเริ่มต้น: 2x)### Anti-thundering herd +Provider profiles support these settings: -เมื่อคำขอหลายรายการส่งถึงผู้ให้บริการที่จำกัดอัตรา OmniRoute จะใช้ mutex + การจำกัดอัตราอัตโนมัติเพื่อซีเรียลไลซ์คำขอและป้องกันความล้มเหลวแบบเรียงซ้อน นี่เป็นการดำเนินการอัตโนมัติสำหรับผู้ให้บริการคีย์ API--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -ผู้ใช้ OmniRoute บางรายวางเกตเวย์ไว้ด้านหน้า RAG หรือสแต็กตัวแทน ในการตั้งค่าเหล่านั้น เป็นเรื่องปกติที่จะเห็นรูปแบบแปลกๆ: OmniRoute ดูเหมาะสมดี (ผู้ให้บริการใช้งาน โปรไฟล์การกำหนดเส้นทางใช้ได้ ไม่มีการแจ้งเตือนจำกัดอัตรา) แต่คำตอบสุดท้ายยังคงผิด +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -ในทางปฏิบัติ เหตุการณ์เหล่านี้มักจะมาจากไปป์ไลน์ RAG ดาวน์สตรีม ไม่ใช่จากเกตเวย์เอง +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -หากคุณต้องการใช้คำศัพท์ร่วมกันเพื่ออธิบายความล้มเหลวเหล่านั้น คุณสามารถใช้ WFGY ProblemMap ซึ่งเป็นแหล่งข้อมูลข้อความใบอนุญาต MIT ภายนอกที่กำหนดรูปแบบความล้มเหลว RAG / LLM ที่เกิดซ้ำสิบหกรูปแบบ ในระดับสูงครอบคลุมถึง: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- การดึงข้อมูลดริฟท์และขอบเขตบริบทที่แตกหัก -- ดัชนีว่างหรือเก่าและร้านค้าเวกเตอร์ -- การฝังกับความหมายที่ไม่ตรงกัน -- ปัญหาแอสเซมบลีและหน้าต่างบริบทพร้อมท์ -- ตรรกะล่มสลายและคำตอบที่มั่นใจมากเกินไป -- ความล้มเหลวในการประสานงานสายโซ่ยาวและตัวแทน -- หน่วยความจำหลายตัวแทนและการเลื่อนบทบาท -- ปัญหาการใช้งานและการสั่งซื้อบูตสแตรป +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -แนวคิดนั้นง่าย: +The idea is simple: -1. เมื่อคุณตรวจสอบการตอบสนองที่ไม่ดี ให้จับ: - - งานของผู้ใช้และการร้องขอ - - เส้นทางหรือคำสั่งผสมผู้ให้บริการใน OmniRoute - - บริบท RAG ใดๆ ที่ใช้ดาวน์สตรีม (เอกสารที่ดึงมา การเรียกเครื่องมือ ฯลฯ) -2. แมปเหตุการณ์กับหมายเลข WFGY ProblemMap หนึ่งหรือสองหมายเลข (`No.1` … `No.16`) -3. จัดเก็บหมายเลขไว้ในแดชบอร์ด รันบุ๊ก หรือตัวติดตามเหตุการณ์ของคุณเอง ถัดจากบันทึก OmniRoute -4. ใช้หน้า WFGY ที่เกี่ยวข้องเพื่อตัดสินใจว่าคุณจำเป็นต้องเปลี่ยนสแต็ก RAG ตัวดึงข้อมูล หรือกลยุทธ์การกำหนดเส้นทางหรือไม่ +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -ข้อความแบบเต็มและสูตรที่เป็นรูปธรรมอยู่ที่นี่ (ใบอนุญาต MIT ข้อความเท่านั้น): +Full text and concrete recipes live here (MIT license, text only): [WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -คุณสามารถเพิกเฉยต่อส่วนนี้ได้หากคุณไม่ได้เรียกใช้ RAG หรือไปป์ไลน์ของตัวแทนที่อยู่เบื้องหลัง OmniRoute--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**ปัญหา GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**สถาปัตยกรรม**: ดู [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) สำหรับรายละเอียดภายใน -**การอ้างอิง API**: ดู [`docs/API_REFERENCE.md`](API_REFERENCE.md) สำหรับตำแหน่งข้อมูลทั้งหมด -**แดชบอร์ดสุขภาพ**: ตรวจสอบ**แดชบอร์ด → สุขภาพ**เพื่อดูสถานะของระบบแบบเรียลไทม์ -**นักแปล**: ใช้**แดชบอร์ด → นักแปล**เพื่อแก้ไขปัญหาเกี่ยวกับรูปแบบ +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/th/llm.txt b/docs/i18n/th/llm.txt new file mode 100644 index 0000000000..a1f508377c --- /dev/null +++ b/docs/i18n/th/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (ไทย) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## ภาพรวม + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### ความปลอดภัย +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/tr/README.md b/docs/i18n/tr/README.md index ce5a6b3e51..1ce855b6ff 100644 --- a/docs/i18n/tr/README.md +++ b/docs/i18n/tr/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Evrensel API proxy'niz — tek uç nokta, 60'tan fazla sağlayıcı, sıfır kesinti süresi. Artık**MCP Sunucusu (25 araç)**,**A2A Protokolü**,**Bellek/Beceri Sistemleri**ve**Electron Masaüstü Uygulaması**._ ile birlikte +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Sohbet Tamamlamaları • Yerleştirmeler • Görüntü Oluşturma • Video • Müzik • Ses • Yeniden Sıralama •**Web Araması**• MCP Sunucusu • A2A Protokolü • %100 TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Evrensel API proxy'niz — tek uç nokta, 60'tan fazla sağlayıcı, sıfır ke [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Web sitesi](https://omniroute.online) • [🚀 Hızlı Başlangıç](#-quick-start) • [💡 Özellikler](#-key-features) • [📖 Dokümanlar](#-documentation) • [💰 Fiyatlandırma](#-bir bakışta fiyatlandırma) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Mevcut dil:**🇺🇸 [İngilizce](README.md) | 🇧🇷 [Portekizce (Brezilya)](docs/i18n/pt-BR/README.md) | 🇪🇸 [İspanyolca](docs/i18n/es/README.md) | 🇫🇷 [Fransızca](docs/i18n/fr/README.md) | 🇮🇹 [İtalyanca](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Almanca](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Endonezya](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Portekiz (Portekiz)](docs/i18n/pt/README.md) | 🇷🇴 [Romanya](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipince](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,553 +60,629 @@ _Evrensel API proxy'niz — tek uç nokta, 60'tan fazla sağlayıcı, sıfır ke ## 📸 Dashboard Preview - -Kontrol paneli ekran görüntülerini görmek için tıklayın +
+Click to see dashboard screenshots -| Sayfa | Ekran görüntüsü | -| ----------------------- | -------------------------------------------------- | ---------- | -| **Sağlayıcılar** | ![Sağlayıcılar](docs/screenshots/01-providers.png) | -| **Kombolar** | ![Kombolar](docs/screenshots/02-combos.png) | -| **Analiz** | ![Analytics](docs/screenshots/03-analytics.png) | -| **Sağlık** | ![Sağlık](docs/screenshots/04-health.png) | -| **Çevirmen** | ![Çevirmen](docs/screenshots/05-translator.png) | -| **Ayarlar** | ![Ayarlar](docs/screenshots/06-settings.png) | -| **CLI Araçları** | ![CLI Araçları](docs/screenshots/07-cli-tools.png) | -| **Kullanım Günlükleri** | ![Kullanım](docs/screenshots/08-usage.png) | -| **Uç noktalar** | ![Uç Noktalar](docs/screenshots/09-endpoint.png) |
| +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | + + --- ### 🤖 Free AI Provider for your favorite coding agents -_Sınırsız kodlama için ücretsiz API ağ geçidi olan OmniRoute aracılığıyla yapay zeka destekli herhangi bir IDE veya CLI aracına bağlanın._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ - + - - - - - - - - - -
+ - OpenClaw
- AçıkPençe + OpenClaw
+ OpenClaw

⭐ 205K
+ - NanoBot
+ NanoBot
NanoBot

- ⭐ 20,9K + ⭐ 20.9K
+ - PicoClaw
+ PicoClaw
PicoClaw

- ⭐ 14,6K + ⭐ 14.6K
+ - ZeroClaw
+ ZeroClaw
ZeroClaw

- ⭐ 9,9K + ⭐ 9.9K
+ - IronClaw
- DemirPençe + IronClaw
+ IronClaw

- ⭐ 2,1K + ⭐ 2.1K
+ - OpenCode
- Açık Kod + OpenCode
+ OpenCode

⭐ 106K
+ - Codex CLI
- Kodeks CLI + Codex CLI
+ Codex CLI

- ⭐ 60,8K + ⭐ 60.8K
+ - Claude Kodu
- Claude Kodu + Claude Code
+ Claude Code

- ⭐ 67,3K + ⭐ 67.3K
+ - Gemini CLI
+ Gemini CLI
Gemini CLI

- ⭐ 94,7K + ⭐ 94.7K
+ - Kilo Kodu
- Kilo Kodu + Kilo Code
+ Kilo Code

- ⭐ 15,5K + ⭐ 15.5K
-📡 Tüm aracılar http://localhost:20128/v1 veya http://cloud.omniroute.online/v1 aracılığıyla bağlanır — tek yapılandırma, sınırsız model ve kota--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Paranızı boşa harcamayı ve limitlere ulaşmayı bırakın:** +**Stop wasting money and hitting limits:** -- Abonelik kotası kullanılmadan her ay sona erer -- Hız sınırları kodlamanın ortasında sizi durdurur -- Pahalı API'ler (sağlayıcı başına ayda 20-50 ABD doları) -- Sağlayıcılar arasında manuel geçiş +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute bunu çözer:** +**OmniRoute solves this:** -- ✅**Abonelikleri en üst düzeye çıkarın**- Kotayı takip edin, sıfırlamadan önce her biti kullanın -- ✅**Otomatik geri dönüş**- Abonelik → API Anahtarı → Ucuz → Ücretsiz, sıfır kesinti -- ✅**Çoklu hesap**- Sağlayıcı başına hesaplar arasında dönüşümlü işlem -- ✅**Evrensel**- Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw ve herhangi bir CLI aracıyla çalışır--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Topluluğumuza katılın!**[WhatsApp Grubu](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Yardım alın, ipuçlarını paylaşın ve güncel kalın. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Web sitesi**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Sorunlar**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Topluluk Grubu](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Katkıda bulunma**: [CONTRIBUTING.md](CONTRIBUTING.md)'ye bakın, bir PR açın veya 'iyi bir ilk sayı' seçin -**Orijinal Proje**: [9router by decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Bir sayıyı açarken lütfen sistem bilgisi komutunu çalıştırın ve oluşturulan dosyayı ekleyin:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Bu, Node.js sürümünüz, OmniRoute sürümünüz, işletim sistemi ayrıntılarınız, yüklü CLI araçlarınız (qoder, gemini, Claude, codex, antigravity, droid vb.), Docker/PM2 durumunuz ve sistem paketlerinizle birlikte bir "system-info.txt" oluşturur; sorununuzu hızlı bir şekilde yeniden oluşturmak için ihtiyacımız olan her şey. Dosyayı doğrudan GitHub sorununuza ekleyin.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Yapay zeka araçlarını kullanan her geliştirici bu sorunlarla her gün karşılaşır.**OmniRoute, maliyet aşımlarından bölgesel engellemelere, bozuk OAuth akışlarından protokol işlemlerine ve kurumsal gözlemlenebilirliğe kadar bunların hepsini çözmek için tasarlandı. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. - -💸 1. "Pahalı bir abonelik için para ödüyorum ama yine de limitler nedeniyle kesintiye uğruyorum" +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Geliştiriciler Claude Pro, Codex Pro veya GitHub Copilot için ayda 20-200 ABD doları ödüyor. Ödeme yaparken bile kotanın bir tavanı vardır: 5 saatlik kullanım, haftalık limitler veya dakika başına ücret limitleri. Kodlama oturumunun ortasında sağlayıcı yanıt vermeyi durdurur ve geliştirici akış ve üretkenliği kaybeder. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**OmniRoute bu sorunu nasıl çözüyor:** +**How OmniRoute solves it:** --**Akıllı 4 Katmanlı Geri Dönüş**— Abonelik kotası biterse, sıfır manuel müdahaleyle otomatik olarak API Anahtarı → Ucuz → Ücretsiz seçeneğine yönlendirilir --**Sağlayıcı Sınırları Takibi**— Önbelleğe alınmış kota anlık görüntüleri, kullanıcı arayüzünde manuel yenileme özelliğiyle sunucu tarafı bir programa göre (varsayılan `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) yenilenir --**Çoklu Hesap Desteği**— Otomatik hepsini bir kez deneme özelliğiyle sağlayıcı başına birden fazla hesap — biri bittiğinde diğerine geçiş yapılır --**Özel Kombinasyonlar**— 9 dengeleme stratejisiyle özelleştirilebilir geri dönüş zincirleri (öncelik, ağırlıklı, önce doldur, hepsini bir kez deneme, P2C, rastgele, en az kullanılan, maliyeti optimize edilmiş, katı rastgele) --**Codex Business Kotaları**— İşletme/Ekip çalışma alanı kotasını doğrudan kontrol panelinden izleme
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard - -🔌 2. "Birden fazla sağlayıcı kullanmam gerekiyor ancak her birinin farklı bir API'si var" + -OpenAI bir format kullanıyor, Claude (Anthropic) başka bir format kullanıyor, Gemini ise bir başka format kullanıyor. Bir geliştirici, farklı sağlayıcıların modellerini test etmek veya bunlar arasında geri dönüş yapmak isterse, SDK'ları yeniden yapılandırması, uç noktaları değiştirmesi ve uyumsuz formatlarla uğraşması gerekir. Özel sağlayıcıların (FriendLI, NIM) standart dışı model uç noktaları vardır. +
+🔌 2. "I need to use multiple providers but each has a different API" -**OmniRoute bu sorunu nasıl çözüyor:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Birleşik Uç Nokta**— Tek bir "http://localhost:20128/v1", 60'tan fazla sağlayıcının tümü için proxy görevi görür --**Biçim Çevirisi**— Otomatik ve şeffaf: OpenAI ↔ Claude ↔ Gemini ↔ Responses API --**Response Temizleme**— OpenAI SDK v1.83+'ı bozan standart olmayan alanları (`x_groq`, `usage_breakdown`, `service_tier`) çıkarır --**Rol Normalleştirme**— OpenAI olmayan sağlayıcılar için "geliştirici" → "sistem"i dönüştürür; GLM/ERNIE için "sistem" → "kullanıcı" --**Think Tag Extraction**— DeepSeek R1 gibi modellerden "" bloklarını standartlaştırılmış "reasoning_content"e çıkarır --**Gemini için Yapılandırılmış Çıktı**— `json_schema` → `responseMimeType`/`responseSchema` otomatik dönüştürme --**`stream` varsayılan olarak `false` olur**— Python/Rust/Go SDK'larında beklenmeyen SSE'yi önleyerek OpenAI spesifikasyonuyla uyumlu hale gelir
+**How OmniRoute solves it:** - -🌐 3. "Yapay zeka sağlayıcım bölgemi/ülkemi engelliyor" +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -OpenAI/Codex gibi sağlayıcılar belirli coğrafi bölgelerden erişimi engelliyor. Kullanıcılar, OAuth ve API bağlantıları sırasında "desteklenmeyen_ülke_bölge_territory" gibi hatalar alıyor. Bu özellikle gelişmekte olan ülkelerdeki geliştiriciler için sinir bozucudur. + -**OmniRoute bu sorunu nasıl çözüyor:** +
+🌐 3. "My AI provider blocks my region/country" --**3 Düzeyli Proxy Yapılandırması**— 3 düzeyde yapılandırılabilir proxy: genel (tüm trafik), sağlayıcı başına (yalnızca bir sağlayıcı) ve bağlantı/anahtar başına --**Renk Kodlu Proxy Rozetleri**— Görsel göstergeler: 🟢 genel proxy, 🟡 sağlayıcı proxy, 🔵 bağlantı proxy'si, her zaman IP'yi gösterir --**Proxy Aracılığıyla OAuth Token Değişimi**— OAuth akışı da proxy üzerinden geçerek "unsupported_country_region_territory" sorununu çözer --**Proxy Aracılığıyla Bağlantı Testleri**— Bağlantı testleri yapılandırılmış proxy'yi kullanır (artık doğrudan geçiş yok) --**SOCKS5 Desteği**— Giden yönlendirme için tam SOCKS5 proxy desteği --**TLS Parmak İzi Sahtekarlığı**— Bot tespitini atlamak için "wreq-js" aracılığıyla tarayıcı benzeri TLS parmak izi --**🔏 CLI Parmak İzi Eşleştirme**— Başlıkları ve gövde alanlarını yerel CLI ikili imzalarıyla eşleşecek şekilde yeniden sıralayarak hesap işaretleme riskini büyük ölçüde azaltır. Proxy IP'si korunur; aynı anda hem gizli**hem de**IP maskeleme elde edersiniz
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. - -🆓 4. "Kodlama için yapay zeka kullanmak istiyorum ama param yok" +**How OmniRoute solves it:** -AI abonelikleri için herkes ayda 20-200 ABD Doları ödeyemez. Öğrencilerin, gelişmekte olan ülkelerdeki geliştiricilerin, amatörlerin ve serbest çalışanların kaliteli modellere sıfır maliyetle erişmesi gerekiyor. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**OmniRoute bu sorunu nasıl çözüyor:** + --**Yerleşik Ücretsiz Katman Sağlayıcıları**— %100 ücretsiz sağlayıcılar için yerel destek: Qoder (OAuth aracılığıyla 5 sınırsız model: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 sınırsız model: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vizyon-model), Kiro (ücretsiz Claude + AWS Builder ID), Gemini CLI (ayda 180.000 jeton ücretsiz) --**Ollama Cloud**— "api.ollama.com" adresinde ücretsiz "Hafif kullanım" katmanıyla bulutta barındırılan Ollama modelleri; 'ollamacloud/' önekini kullan --**Yalnızca Ücretsiz Kombinasyonlar**— Zincir `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = sıfır kesinti süresiyle ayda 0 ABD doları --**NVIDIA NIM Ücretsiz Erişim**— ~40 RPM geliştirici-build.nvidia.com adresinde 70'ten fazla modele sonsuza kadar ücretsiz erişim (kredilerden saf hız sınırlarına geçiş) --**Maliyet Optimize Edilmiş Strateji**— Mevcut en ucuz sağlayıcıyı otomatik olarak seçen yönlendirme stratejisi +
+🆓 4. "I want to use AI for coding but I have no money" - -🔒 5. "Yapay zeka ağ geçidimi yetkisiz erişime karşı korumam gerekiyor" +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -Bir AI ağ geçidini ağa (LAN, VPS, Docker) sunarken, adrese sahip olan herkes geliştiricinin belirteçlerini/kotasını kullanabilir. Koruma olmadan API'ler kötüye kullanıma, anında enjeksiyona ve kötüye kullanıma karşı savunmasızdır. +**How OmniRoute solves it:** -**OmniRoute bu sorunu nasıl çözüyor:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**API Anahtar Yönetimi**— Özel bir "/dashboard/api-manager" sayfasıyla sağlayıcı başına oluşturma, döndürme ve kapsam belirleme --**Model Düzeyinde İzinler**— Tümüne İzin Ver/Kısıtla geçişiyle API anahtarlarını belirli modellerle ('openai/*', joker karakter desenleri) kısıtlayın --**API Uç Nokta Koruması**— `/v1/models' için bir anahtar zorunlu kılın ve belirli sağlayıcıların listede yer almasını engelleyin --**Auth Guard + CSRF Koruması**— Tüm kontrol paneli rotaları "withAuth" ara yazılımı + CSRF belirteçleriyle korunur --**Hız Sınırlayıcı**— Yapılandırılabilir pencerelerle IP başına hız sınırlaması --**IP Filtreleme**— Erişim kontrolü için izin verilenler listesi/engellenenler listesi --**Prompt Injection Guard**— Kötü niyetli istem kalıplarına karşı temizleme --**AES-256-GCM Şifreleme**— Kullanımda değilken şifrelenen kimlik bilgileri
+ - -🛑 6. "Sağlayıcım arızalandı ve kodlama akışımı kaybettim" +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -Yapay zeka sağlayıcıları kararsız hale gelebilir, 5xx hataları döndürebilir veya geçici hız sınırlarına ulaşabilir. Bir geliştirici tek bir sağlayıcıya bağlıysa kesintiye uğrar. Devre kesiciler olmadan tekrarlanan yeniden denemeler uygulamanın çökmesine neden olabilir. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**OmniRoute bu sorunu nasıl çözüyor:** +**How OmniRoute solves it:** --**Model başına Devre Kesici**— Yapılandırılabilir eşikler ve bekleme süresi (Kapalı/Açık/Yarı Açık) ile otomatik açma/kapama, basamaklı blokları önlemek için model başına kapsamlıdır --**Üstel Gerileme**— Aşamalı yeniden deneme gecikmeleri --**Gök Gürültüsü Önleyici Sürü**— Eş zamanlı yeniden deneme fırtınalarına karşı Mutex + semafor koruması --**Birleşik Geri Dönüş Zincirleri**— Birincil sağlayıcı başarısız olursa, hiçbir müdahale olmadan otomatik olarak zincirden düşer --**Kombo Devre Kesici**— Birleşik zincir içindeki arızalı sağlayıcıları otomatik olarak devre dışı bırakır --**Sağlık Kontrol Paneli**— Çalışma süresi izleme, devre kesici durumları, kilitlenmeler, önbellek istatistikleri, p50/p95/p99 gecikmesi
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest - -🔧 7. "Her bir AI aracını yapılandırmak yorucu ve tekrarlayıcıdır" + -Geliştiriciler Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Her aracın farklı bir yapılandırmaya (API uç noktası, anahtar, model) ihtiyacı vardır. Sağlayıcıları veya modelleri değiştirirken yeniden yapılandırmak zaman kaybıdır. +
+🛑 6. "My provider went down and I lost my coding flow" -**OmniRoute bu sorunu nasıl çözüyor:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**CLI Araçları Kontrol Paneli**— Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline için tek tıklamayla kurulum sağlayan özel sayfa --**GitHub Copilot Config Generator**— Toplu model seçimiyle VS Kodu için `chatLanguageModels.json` oluşturur --**İlk Katılım Sihirbazı**— İlk kez kullananlar için kılavuzlu 4 adımlı kurulum --**Tek uç nokta, tüm modeller**— `http://localhost:20128/v1`'i bir kez yapılandırın, 60'tan fazla sağlayıcıya erişin
+**How OmniRoute solves it:** - -🔑 8. "Birden fazla sağlayıcıdan gelen OAuth jetonlarını yönetmek cehennemdir" +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot — hepsi süresi dolan jetonlarla OAuth 2.0 kullanıyor. Geliştiricilerin sürekli olarak yeniden kimlik doğrulaması yapması, "client_secret eksik", "redirect_uri_mismatch" ve uzak sunuculardaki arızalarla uğraşması gerekir. LAN/VPS'de OAuth özellikle sorunludur. + -**OmniRoute bu sorunu nasıl çözüyor:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Otomatik Belirteç Yenileme**— OAuth belirteçleri, geçerlilik süresi dolmadan arka planda yenilenir --**OAuth 2.0 (PKCE) Yerleşik**— Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder için otomatik akış --**Çoklu Hesap OAuth**— JWT/ID belirteci çıkarma yoluyla sağlayıcı başına birden fazla hesap --**OAuth LAN/Remote Fix**— "redirect_uri" için özel IP algılama + uzak sunucular için manuel URL modu --**Nginx'in Arkasında OAuth**— Ters proxy uyumluluğu için `window.location.origin'i kullanır --**Uzaktan OAuth Kılavuzu**— VPS/Docker'da Google Cloud kimlik bilgileri için adım adım kılavuz
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. - -📊 9. "Ne kadar harcadığımı veya nereye harcadığımı bilmiyorum" +**How OmniRoute solves it:** -Geliştiriciler birden fazla ücretli sağlayıcı kullanıyor ancak harcamalara ilişkin birleşik bir görünüme sahip değiller. Her sağlayıcının kendi faturalandırma kontrol paneli vardır ancak birleştirilmiş görünüm yoktur. Beklenmedik maliyetler birikebilir. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**OmniRoute bu sorunu nasıl çözüyor:** + --**Maliyet Analitiği Kontrol Paneli**— Sağlayıcı başına jeton başına maliyet takibi ve bütçe yönetimi --**Kademe Başına Bütçe Sınırları**— Otomatik geri dönüşü tetikleyen, kademe başına harcama tavanı --**Model Başına Fiyatlandırma Yapılandırması**— Model başına yapılandırılabilir fiyatlar --**API Anahtarı Başına Kullanım İstatistikleri**— Anahtar başına istek sayısı ve son kullanılan zaman damgası --**Analytics Kontrol Paneli**— İstatistik kartları, model kullanım tablosu, başarı oranlarını ve gecikmeyi gösteren sağlayıcı tablosu +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" - -🐛 10. "Yapay zeka çağrılarındaki hataları ve sorunları teşhis edemiyorum" +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Bir çağrı başarısız olduğunda geliştirici bunun bir hız limiti mi, süresi dolmuş bir jeton mu, yanlış format mı yoksa sağlayıcı hatası mı olduğunu bilemez. Farklı terminallerde parçalanmış günlükler. Gözlemlenebilirlik olmadığında hata ayıklama deneme yanılma yöntemiyle yapılır. +**How OmniRoute solves it:** -**OmniRoute bu sorunu nasıl çözüyor:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Birleşik Günlük Kontrol Paneli**— 4 sekme: İstek Günlükleri, Proxy Günlükleri, Denetim Günlükleri, Konsol --**Konsol Günlük Görüntüleyici**— Renk kodlu seviyelere, otomatik kaydırmaya, arama ve filtreye sahip gerçek zamanlı terminal tarzı görüntüleyici --**SQLite Proxy Günlükleri**— Sunucu yeniden başlatıldığında hayatta kalan kalıcı günlükler --**Çevirmen Oyun Alanı**— 4 hata ayıklama modu: Oyun Alanı (biçim çevirisi), Sohbet Test Cihazı (gidiş-dönüş), Test Cihazı (toplu), Canlı Monitör (gerçek zamanlı) --**İstek Telemetrisi**— p50/p95/p99 gecikmesi + X-İstek Kimliği izleme --**Döndürmeli Dosya Tabanlı Günlük Kaydı**— Uygulama günlükleri boyuta, saklama günlerine ve arşiv sayısına göre değişir; çağrı günlüğü yapıları, saklama günlerine ve dosya sayısına göre değişir --**Sistem Bilgisi Raporu**— "npm run system-info", tüm ortamınızla (Node sürümü, OmniRoute sürümü, işletim sistemi, CLI araçları, Docker/PM2 durumu) "system-info.txt" dosyasını oluşturur. Anında önceliklendirme için sorunları bildirirken bunu ekleyin.
+ - -🏗️ 11. "Ağ geçidini dağıtmak ve sürdürmek karmaşıktır" +
+📊 9. "I don't know how much I'm spending or where" -Farklı ortamlarda (yerel, VPS, Docker, bulut) bir AI proxy'yi kurmak, yapılandırmak ve sürdürmek yoğun emek gerektirir. Sabit kodlanmış yollar, dizinlerdeki "EACCES", bağlantı noktası çakışmaları ve platformlar arası yapılar gibi sorunlar sürtüşmeyi artırır. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**OmniRoute bu sorunu nasıl çözüyor:** +**How OmniRoute solves it:** --**npm global kurulumu**— `npm kurulumu -g omniroute && omniroute` — tamamlandı --**Docker Çoklu Platform**— AMD64 + ARM64 yerel (Apple Silicon, AWS Graviton, Raspberry Pi) --**Docker Compose Profilleri**— "temel" (CLI aracı yok) ve "cli" (Claude Code, Codex, OpenClaw ile) --**Electron Masaüstü Uygulaması**— Windows/macOS/Linux için sistem tepsisi, otomatik başlatma, çevrimdışı mod içeren yerel uygulama --**Bölünmüş Bağlantı Noktası Modu**— Gelişmiş senaryolar için ayrı bağlantı noktalarında API ve Kontrol Paneli (ters proxy, konteyner ağı) --**Bulut Senkronizasyonu**— Cloudflare Workers aracılığıyla cihazlar arasında senkronizasyonu yapılandırın --**DB Yedeklemeleri**— Harici olarak yönetilen yedeklemeler için "DISABLE_SQLITE_AUTO_BACKUP" ile tüm ayarların otomatik olarak yedeklenmesi, geri yüklenmesi, dışa ve içe aktarılması
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency - -🌍 12. "Arayüz yalnızca İngilizcedir ve ekibim İngilizce konuşmuyor" + -İngilizce konuşulmayan ülkelerdeki, özellikle Latin Amerika, Asya ve Avrupa'daki takımlar, yalnızca İngilizce arayüzlerde sorun yaşıyor. Dil engelleri benimsemeyi azaltır ve yapılandırma hatalarını artırır. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**OmniRoute bu sorunu nasıl çözüyor:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Dashboard i18n — 30 Dil**— Arapça, Bulgarca, Danca, Almanca, İspanyolca, Fince, Fransızca, İbranice, Hintçe, Macarca, Endonezce, İtalyanca, Japonca, Korece, Malayca, Felemenkçe, Norveççe, Lehçe, Portekizce (PT/BR), Rumence, Rusça, Slovakça, İsveççe, Tayca, Ukraynaca, Vietnamca, Çince, Filipince, İngilizce dahil olmak üzere 500'den fazla anahtarın tümü çevrildi --**RTL Desteği**— Arapça ve İbranice için sağdan sola destek --**Çok Dilli README'ler**— 30 tam belge çevirisi --**Dil Seçici**— Gerçek zamanlı geçiş için başlıktaki küre simgesi
+**How OmniRoute solves it:** - -🔄 13. "Sohbetten daha fazlasına ihtiyacım var; yerleştirmelere, resimlere ve sese ihtiyacım var" +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -Yapay zeka yalnızca sohbetin tamamlanması değildir. Geliştiricilerin görüntüler oluşturması, sesi yazıya dökmesi, RAG için yerleştirmeler oluşturması, belgeleri yeniden sıralaması ve içeriği denetlemesi gerekiyor. Her API'nin farklı bir uç noktası ve biçimi vardır. + -**OmniRoute bu sorunu nasıl çözüyor:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Yerleştirmeler**— 6 sağlayıcı ve 9'dan fazla modelle `/v1/yerleştirmeler` --**Görüntü Oluşturma**— 10 sağlayıcı ve 20'den fazla modelle (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolik, NanoBanana, Antigravity, SD WebUI, ComfyUI) `/v1/images/ Generations` --**Metinden Videoya**— `/v1/videos/ Generations` — ComfyUI (AnimateDiff, SVD) ve SD WebUI --**Metinden Müziğe**— `/v1/music/ Generations` — ComfyUI (Stable Audio Open, MusicGen) --**Ses Transkripsiyonu**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Metin-Konuşma**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + mevcut sağlayıcılar --**Denetlemeler**— `/v1/moderations` — İçerik güvenliği kontrolleri --**Yeniden sıralama**— `/v1/rerank` — Belge alaka düzeyinin yeniden sıralaması --**Responses API**— Codex için tam `/v1/responses` desteği
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. - -🧪 14. "Kaliteyi modeller arasında test etme ve karşılaştırma yöntemim yok" +**How OmniRoute solves it:** -Geliştiriciler kendi kullanım durumları için hangi modelin (kod, çeviri, akıl yürütme) en iyi olduğunu bilmek isterler ancak manuel olarak karşılaştırma yapmak yavaştır. Entegre değerlendirme aracı mevcut değildir. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**OmniRoute bu sorunu nasıl çözüyor:** + --**LLM Değerlendirmeleri**— Selamlama, matematik, coğrafya, kod oluşturma, JSON uyumluluğu, çeviri, indirim, güvenlik reddini kapsayan önceden yüklenmiş 10 vakayla altın set testi --**4 Eşleşme Stratejisi**— "tam", "içerir", "regex", "özel" (JS işlevi) --**Translator Playground Test Bench**— Çoklu giriş ve beklenen çıktılarla toplu test, sağlayıcılar arası karşılaştırma --**Sohbet Test Cihazı**— Görsel yanıt oluşturmayla tam gidiş-dönüş --**Canlı İzleme**— Proxy üzerinden akan tüm isteklerin gerçek zamanlı akışı +
+🌍 12. "The interface is English-only and my team doesn't speak English" - -📈 15. "Performansı kaybetmeden ölçeklendirmem gerekiyor" +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -İstek hacmi büyüdükçe, aynı soruları önbelleğe almamak, mükerrer maliyetlere neden olur. Belirsizlik olmadan, yinelenen istekler atık işlemeyi gerektirir. Sağlayıcı başına ücret sınırlarına uyulmalıdır. +**How OmniRoute solves it:** -**OmniRoute bu sorunu nasıl çözüyor:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Anlamsal Önbellek**— İki katmanlı önbellek (imza + anlamsal), maliyeti ve gecikmeyi azaltır --**Talep Idempotency**— Aynı istekler için 5 saniyelik veri tekilleştirme penceresi --**Hız Limiti Tespiti**— Sağlayıcı başına RPM, minimum aralık ve maksimum eşzamanlı izleme --**Düzenlenebilir Hız Sınırları**— Ayarlar'da yapılandırılabilir varsayılanlar → Kalıcı dayanıklılık --**API Anahtar Doğrulama Önbelleği**— Üretim performansı için 3 katmanlı önbellek --**Telemetri Özellikli Sağlık Kontrol Paneli**— p50/p95/p99 gecikmesi, önbellek istatistikleri, çalışma süresi
+ - -🤖 16. "Model davranışını genel olarak kontrol etmek istiyorum" +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Tüm yanıtların belirli bir dilde, belirli bir tonda olmasını isteyen veya akıl yürütme belirteçlerini sınırlamak isteyen geliştiriciler. Bunu her araçta/istekte yapılandırmak pratik değildir. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**OmniRoute bu sorunu nasıl çözüyor:** +**How OmniRoute solves it:** --**Sistem İstemi Ekleme**— Tüm isteklere genel istem uygulandı --**Bütçe Doğrulamasını Düşünme**— İstek başına jeton tahsis kontrolünün akıl yürütmesi (geçişli, otomatik, özel, uyarlanabilir) --**9 Yönlendirme Stratejileri**— İsteklerin nasıl dağıtılacağını belirleyen genel stratejiler --**Wildcard Router**— `sağlayıcı/*' kalıpları herhangi bir sağlayıcıya dinamik olarak yönlendirir --**Kombo Etkinleştirme/Devre Dışı Geçişi**— Kombinasyonları doğrudan kontrol panelinden değiştirin --**Sağlayıcı Geçişi**— Bir sağlayıcı için tüm bağlantıları tek tıklamayla etkinleştirin/devre dışı bırakın --**Engellenen Sağlayıcılar**— Belirli sağlayıcıları `/v1/models` listesinden hariç tutun
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex - -🧰 17. "Birinci sınıf ürün özellikleri olarak MCP araçlarına ihtiyacım var" + -Birçok AI ağ geçidi, MCP'yi yalnızca gizli bir uygulama ayrıntısı olarak açığa çıkarır. Ekiplerin görünür ve yönetilebilir bir operasyon katmanına ihtiyacı vardır. +
+🧪 14. "I have no way to test and compare quality across models" -**OmniRoute bu sorunu nasıl çözüyor:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP, kontrol panelinde gezinme ve uç nokta protokolü sekmesinde görünür -- Süreç, araçlar, kapsamlar ve denetimin yer aldığı özel MCP yönetimi sayfası -- 'omniroute --mcp' ve istemci katılımı için yerleşik hızlı başlangıç
+**How OmniRoute solves it:** - -🧠 18. "Senkronizasyon + akış görev yollarıyla A2A orkestrasyonuna ihtiyacım var" +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Aracı iş akışları, hem doğrudan yanıtlara hem de yaşam döngüsü kontrolüyle uzun süreli akışlı yürütmeye ihtiyaç duyar. + -**OmniRoute bu sorunu nasıl çözüyor:** +
+📈 15. "I need to scale without losing performance" -- "mesaj/gönder" ve "mesaj/akış" ile A2A JSON-RPC uç noktası ("POST /a2a") -- Terminal durumu yayılımıyla SSE akışı -- "görevler/al" ve "görevler/iptal" için görev yaşam döngüsü API'leri
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. - -🛰️ 19. "Tahmin edilen duruma değil, gerçek MCP süreç durumuna ihtiyacım var" +**How OmniRoute solves it:** -Operasyonel ekiplerin yalnızca bir API'nin erişilebilir olup olmadığını değil, MCP'nin gerçekten hayatta olup olmadığını bilmesi gerekir. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**OmniRoute bu sorunu nasıl çözüyor:** + -- PID, zaman damgaları, aktarım, araç sayısı ve kapsam modunu içeren çalışma zamanı kalp atışı dosyası -- Kalp atışı + son etkinliği birleştiren MCP durum API'si -- Süreç/çalışma süresi/kalp atışı tazeliği için kullanıcı arayüzü durum kartları +
+🤖 16. "I want to control model behavior globally" - -📋 20. "Denetlenebilir MCP aracı uygulamasına ihtiyacım var" +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Araçlar yapılandırmayı değiştirdiğinde veya operasyon eylemlerini tetiklediğinde ekiplerin adli izlenebilirliğe ihtiyacı vardır. +**How OmniRoute solves it:** -**OmniRoute bu sorunu nasıl çözüyor:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- MCP aracı çağrıları için SQLite destekli denetim günlüğü -- Araç, başarı/başarısızlık, API anahtarı ve sayfalandırmaya göre filtreler -- Otomasyon için kontrol paneli denetim tablosu + istatistik uç noktaları
+ - -🔐 21. "Entegrasyon başına kapsamlı MCP izinlerine ihtiyacım var" +
+🧰 17. "I need MCP tools as first-class product capabilities" -Farklı istemcilerin araç kategorilerine en az ayrıcalıklı erişime sahip olması gerekir. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**OmniRoute bu sorunu nasıl çözüyor:** +**How OmniRoute solves it:** -- Kontrollü araç erişimi için 10 ayrıntılı MCP kapsamı -- MCP yönetimi kullanıcı arayüzünde kapsam uygulaması ve görünürlük -- Operasyonel takımlama için güvenli varsayılan duruş
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding - -⚙️ 22. "Yeniden dağıtım yapmadan operasyonel kontrollere ihtiyacım var" + -Ekiplerin olaylar veya maliyet olayları sırasında hızlı çalışma süresi değişikliklerine ihtiyacı vardır. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**OmniRoute bu sorunu nasıl çözüyor:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Birleşik aktivasyonu doğrudan MCP kontrol panelinden değiştirin -- Önceden tanımlanmış politika paketlerinden dayanıklılık profillerini uygulayın -- Aynı işlem panelinden devre kesici durumunu sıfırlayın
+**How OmniRoute solves it:** - -🔄 23. "Canlı A2A görevi yaşam döngüsü görünürlüğüne ve iptaline ihtiyacım var" +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Yaşam döngüsü görünürlüğü olmadan görev olaylarının önceliklendirilmesi zorlaşır. + -**OmniRoute bu sorunu nasıl çözüyor:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- Sayfalandırmayla duruma/beceriye göre görev listeleme/filtreleme -- Görev meta verileri, olaylar ve yapılar üzerinde ayrıntılı inceleme -- Onay ile görev iptali uç noktası ve kullanıcı arayüzü eylemi
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. - -🌊 24. "A2A yükü için aktif akış ölçümlerine ihtiyacım var" +**How OmniRoute solves it:** -Akış iş akışları, eşzamanlılık ve canlı bağlantılara ilişkin operasyonel bilgiler gerektirir. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**OmniRoute bu sorunu nasıl çözüyor:** + -- A2A durumuna entegre edilmiş aktif akış sayaçları -- Son görev zaman damgası ve eyalet başına sayımlar -- Gerçek zamanlı operasyonların izlenmesi için A2A kontrol paneli kartları +
+📋 20. "I need auditable MCP tool execution" - -🪪 25. "Müşteriler için standart aracı keşfine ihtiyacım var" +When tools mutate config or trigger ops actions, teams need forensic traceability. -Harici istemciler ve orkestratörlerin katılım için makine tarafından okunabilir meta verilere ihtiyacı vardır. +**How OmniRoute solves it:** -**OmniRoute bu sorunu nasıl çözüyor:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- Temsilci Kartı `/.well-known/agent.json` adresinde gösteriliyor -- Yönetim kullanıcı arayüzünde gösterilen yetenekler ve beceriler -- A2A durum API'si otomasyon için keşif meta verilerini içerir
+ - -🧭 26. "Ürün kullanıcı deneyiminde protokolün keşfedilebilirliğine ihtiyacım var" +
+🔐 21. "I need scoped MCP permissions per integration" -Kullanıcılar protokol yüzeylerini keşfedemezse benimseme ve destek kalitesi düşer. +Different clients should have least-privilege access to tool categories. -**OmniRoute bu sorunu nasıl çözüyor:** +**How OmniRoute solves it:** -- Proxy, MCP, A2A ve API Uç Noktaları sekmelerini içeren birleştirilmiş**Uç Noktalar**sayfası -- MCP ve A2A için hat içi hizmet durumu geçişleri (Çevrimiçi/Çevrimdışı) -- Genel bakıştan özel yönetim sekmelerine bağlantılar
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling - -🧪 27. "Gerçek istemcilerle uçtan uca protokol doğrulamaya ihtiyacım var" + -Deneme testleri, yayınlanmadan önce protokol uyumluluğunu doğrulamak için yeterli değildir. +
+⚙️ 22. "I need operational controls without redeploying" -**OmniRoute bu sorunu nasıl çözüyor:** +Teams need quick runtime changes during incidents or cost events. -- Uygulamayı başlatan ve gerçek MCP SDK istemci aktarımını kullanan E2E paketi -- Akışları keşfetme, gönderme, yayınlama, alma ve iptal etme için A2A istemci testleri -- MCP denetimi ve A2A görevleri API'lerine karşı iddiaları çapraz kontrol edin
+**How OmniRoute solves it:** - -📡 28. "Tüm arayüzlerde birleşik gözlemlenebilirliğe ihtiyacım var" +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -Gözlemlenebilirliğin protokole göre bölünmesi, kör noktalar ve daha uzun MTTR oluşturur. + -**OmniRoute bu sorunu nasıl çözüyor:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Tek bir üründe birleştirilmiş kontrol panelleri/günlükler/analizler -- OpenAI, MCP ve A2A katmanlarında sağlık + denetim + istek telemetrisi -- Durum ve otomasyon için operasyonel API'ler
+Without lifecycle visibility, task incidents become hard to triage. - -💼 29. "Proxy + araçlar + aracı orkestrasyonu için bir çalışma zamanına ihtiyacım var" +**How OmniRoute solves it:** -Birçok ayrı hizmetin çalıştırılması operasyonel maliyeti ve arıza türlerini artırır. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**OmniRoute bu sorunu nasıl çözüyor:** + -- OpenAI uyumlu proxy, MCP sunucusu ve A2A sunucusu tek bir yığında -- Paylaşılan kimlik doğrulama, esneklik, veri deposu ve gözlemlenebilirlik -- Tüm etkileşim yüzeylerinde tutarlı politika modeli +
+🌊 24. "I need active stream metrics for A2A load" - -🚀 30. "Yapıştırıcı kodun yayılması olmadan ajansal iş akışları göndermem gerekiyor" +Streaming workflows require operational insight into concurrency and live connections. -Ekipler birden fazla geçici hizmeti ve komut dosyasını birleştirirken hız kaybeder. +**How OmniRoute solves it:** -**OmniRoute bu sorunu nasıl çözüyor:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Müşteriler ve temsilciler için birleşik uç nokta stratejisi -- Yerleşik protokol yönetimi kullanıcı arayüzleri ve duman doğrulama yolları -- Üretime hazır temeller (güvenlik, günlük kaydı, esneklik, yedekleme)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Başucu Kitabı A: Ücretli aboneliği en üst düzeye çıkarın + ucuz yedekleme**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -607,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Başucu Kitabı B: Sıfır maliyetli kodlama yığını**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Başucu Kitabı C: 7/24 her zaman açık geri dönüş zinciri**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -630,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Başucu Kitabı D: MCP + A2A ile ajan operasyonları**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost ->**0$/ay**karşılığında AI kodlamasını dakikalar içinde kurun. Bu ücretsiz hesapları bağlayın ve yerleşik**Free Stack**kombinasyonunu kullanın. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Adım | Eylem | Sağlayıcıların Kilidi Açıldı | +| Step | Action | Providers Unlocked | | ---- | -------------------------------------------------- | ------------------------------------------------------------------ | -| 1 |**Kiro**'yu bağlayın (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**sınırsız**| -| 2 |**Qoder**'a bağlanın (Google OAuth) | kimi-k2-düşünme, qwen3-kodlayıcı-artı, deepseek-r1... —**sınırsız**| -| 3 |**Qwen**'i bağlayın (Cihaz Kodu) | qwen3-coder-plus, qwen3-coder-flash... —**sınırsız**| -| 4 |**Gemini CLI**'yi (Google OAuth) bağlayın | gemini-3-flash, gemini-2.5-pro —**180K/ay ücretsiz**| -| 5 | `/dashboard/combos` →**Ücretsiz Yığın ($0)**şablon | Tüm ücretsiz sağlayıcıları otomatik olarak sıralayın | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Herhangi bir IDE/CLI'yi şuraya yönlendirin:**`http://localhost:20128/v1` · API Anahtarı: `any-string` · Bitti. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**İsteğe bağlı ekstra kapsam (ayrıca ücretsiz):**Groq API anahtarı (30 RPM ücretsiz), NVIDIA NIM (40 RPM ücretsiz, 70'den fazla model), Cerebras (1 milyon tok/gün), LongCat API anahtarı (50 milyon jeton/gün!), Cloudflare Workers AI (10 bin nöron/gün, 50'den fazla model).## Hızlı Başlangıç +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Hızlı Başlangıç ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **pnpm kullanıcıları:**"better-sqlite3" ve "@swc/core" tarafından gereken yerel derleme komut dosyalarını etkinleştirmek için kurulumdan sonra "pnpm onay-builds -g" komutunu çalıştırın: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > > ```bash > pnpm install -g omniroute -> pnpm onayla-derlemeler -g # Tüm paketleri seç → onayla -> çok yönlü +> pnpm approve-builds -g # Select all packages → approve +> omniroute > ``` -Kontrol paneli "http://localhost:20128" konumunda açılır ve API temel URL'si "http://localhost:20128/v1" olur. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Komut | Açıklama | -| ----------------------------- | -------------------------------------------------------------------------------- | -| 'çok yönlü' | Sunucuyu başlatın (`PORT=20128`, API ve kontrol paneli aynı bağlantı noktasında) | -| 'omniroute --port 3000' | Kurallı/API bağlantı noktasını 3000 olarak ayarlayın | -| 'çok yönlü rota --mcp' | MCP sunucusunu başlatın (stdio aktarımı) | -| `çok yönlü rota --açık değil' | Tarayıcıyı otomatik olarak açma | -| `çok yönlü rota --yardım' | Yardımı göster | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -İsteğe bağlı bölünmüş bağlantı noktası modu:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -Çoğu dağıtım için yalnızca şunlara ihtiyacınız vardır: +For most deployments, you only need: -| Değişken | Varsayılan | Amaç | -| ------------------------ | ----------------------------- | ---------------------------------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | '600000' | Yukarı akış getirme, gizli Undici zaman aşımları, TLS parmak izi istekleri ve API köprüsü isteği/proxy zaman aşımları için paylaşılan temel | -| `STREAM_IDLE_TIMEOUT_MS` | `REQUEST_TIMEOUT_MS`yi devralır | OmniRoute SSE akışını iptal etmeden önce akış parçaları arasındaki maksimum boşluk | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -Geriye dönük uyumluluk korunur: mevcut "FETCH_TIMEOUT_MS", "API_BRIDGE_PROXY_TIMEOUT_MS" ve katman başına diğer zaman aşımı değişkenleri çalışmaya devam eder ve paylaşılan temel çizgiyi geçersiz kılar. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Daha hassas kontrole ihtiyacınız varsa gelişmiş geçersiz kılmalar mevcuttur:| Değişken | Varsayılan | Amaç | -| ---------------------------------------- | ------------------------------- | -------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | `REQUEST_TIMEOUT_MS`yi devralır | Ana alma iptal sinyali tarafından kullanılan toplam yukarı akış isteği zaman aşımı | -| `FETCH_HEADERS_TIMEOUT_MS` | `FETCH_TIMEOUT_MS`yi devralır | Yukarı akış yanıt başlıklarını almak için süre sınırı | -| `FETCH_BODY_TIMEOUT_MS` | `FETCH_TIMEOUT_MS`yi devralır | Yukarı akışlı gövde parçaları arasındaki süre sınırı ("0" bunu devre dışı bırakır) | -| `FETCH_CONNECT_TIMEOUT_MS` | '30000' | TCP bağlantısı zaman aşımına uğradı | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | '4000' | Boşta kalma canlı tutma yuvası zaman aşımı | -| `TLS_CLIENT_TIMEOUT_MS` | `FETCH_TIMEOUT_MS`yi devralır | 'Wreq-js' aracılığıyla yapılan TLS parmak izi istekleri için zaman aşımı | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | 'REQUEST_TIMEOUT_MS' veya '30000' değerini devralır | API bağlantı noktasından kontrol paneli bağlantı noktasına `/v1` proxy iletiminde zaman aşımı | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | API köprü sunucusunda gelen istek zaman aşımı | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | '60000' | API köprü sunucusunda gelen başlık zaman aşımı | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | '5000' | API köprü sunucusunda canlı tutma zaman aşımı | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | '0' | API köprü sunucusunda soket hareketsizliği zaman aşımı ("0" bunu devre dışı bırakır) | +Advanced overrides are available if you need finer control: -OmniRoute'u Nginx, Caddy, Cloudflare veya başka bir ters proxy'nin arkasında çalıştırıyorsanız proxy'nin olduğundan emin olun. -zaman aşımları aynı zamanda OmniRoute akış/getirme zaman aşımlarınızdan da daha yüksektir.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Kontrol Paneli → `Sağlayıcılar'ı açın ve en az bir sağlayıcıya (OAuth veya API anahtarı) bağlanın. -2. Kontrol Paneli → `Uç Noktalar'ı açın ve bir API anahtarı oluşturun. -3. (İsteğe bağlı) Kontrol Paneli → 'Kombolar'ı açın ve yedek zincirinizi ayarlayın.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode ve OpenAI uyumlu SDK'larla çalışır.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (araç odaklı işlemler için):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` +Then connect your MCP client over `stdio` and test tools like: -Daha sonra MCP istemcinizi "stdio" üzerinden bağlayın ve aşağıdaki gibi araçları test edin: +- `omniroute_get_health` +- `omniroute_list_combos` -- "omniroute_get_health" -- "çoklu rota_listesi_birleşimleri" +**A2A (for agent-to-agent workflows):** -**A2A (temsilciden temsilciye iş akışları için):**```bash +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -759,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Bu paket, gerçek MCP ve A2A istemci akışlarını çalışan bir uygulamaya göre doğrular.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -767,13 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` - -Linux'u geçersiz kılın (`xbps-src` şablonu) +
+Void Linux (`xbps-src` template) -Void Linux kullanıcıları için 'xbps-src'yi kullanarak yerel bir paket oluşturabilirsiniz. Bu bloğu `srcpkgs/omniroute/template` olarak kaydedin:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -785,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -793,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -869,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -880,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute, [Docker Hub'ında](https://hub.docker.com/r/diegosouzapw/omniroute) herkese açık bir Docker görüntüsü olarak mevcuttur. +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Hızlı çalışma:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -890,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**Ortam dosyasıyla:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Docker Compose'u kullanma:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Docker dağıtımları için kontrol paneli desteği artık "Kontrol Paneli → Uç Noktalar"da tek tıklamayla**Cloudflare Hızlı Tüneli**içeriyor. İlk etkinleştirme, yalnızca gerektiğinde "cloudflared" indirmelerini etkinleştirir, geçerli "/v1" uç noktanıza geçici bir tünel başlatır ve oluşturulan "https://\*.trycloudflare.com/v1" URL'sini doğrudan normal genel URL'nizin altında gösterir. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Notlar: +Notes: -- Hızlı Tünel URL'leri geçicidir ve her yeniden başlatmanın ardından değişir. -- Hızlı Tüneller, OmniRoute veya konteyner yeniden başlatıldıktan sonra otomatik olarak geri yüklenmez. Gerektiğinde bunları kontrol panelinden yeniden etkinleştirin. -- Yönetilen yükleme şu anda "x64" / "arm64" üzerinde Linux, macOS ve Windows'u desteklemektedir. -- Yönetilen Hızlı Tüneller, kısıtlı kapsayıcı ortamlarında gürültülü QUIC UDP arabellek uyarılarını önlemek için varsayılan olarak HTTP/2 aktarımını kullanır. Farklı bir aktarım istiyorsanız `CLOUDFLARED_PROTOCOL=quic` veya `auto` seçeneğini ayarlayın. -- Docker görüntüleri, sistem CA köklerini paketler ve bunları yönetilen "cloudflared"a iletir; bu, tünel konteynerin içinde önyükleme yaptığında TLS güven hatalarını önler. -- SQLite WAL modunda çalışır. OmniRoute'un "storage.sqlite" dosyasındaki en son değişiklikleri yeniden kontrol edebilmesi için "docker stop"un bitmesine izin verilmelidir. -- Birlikte verilen Compose dosyaları zaten 40 saniyelik bir durdurma yetkisiz kullanım süresi belirlemiştir. Görüntüyü doğrudan çalıştırıyorsanız, `--stop-timeout 40` (veya benzeri) tutun, böylece manuel durdurmalar kapatma temizliğini kesintiye uğratmaz. -- OmniRoute'un bir ikili dosyayı indirmek yerine mevcut bir ikili programı kullanmasını istiyorsanız `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` ayarını yapın. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Docker Compose'u Caddy (HTTPS Auto-TLS) ile kullanma:** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute, Caddy'nin otomatik SSL provizyonu kullanılarak güvenli bir şekilde açığa çıkarılabilir. Alan adınızın DNS A kaydının sunucunuzun IP'sini gösterdiğinden emin olun.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` - -| Resim | Etiket | Boyut | Açıklama | +| Image | Tag | Size | Description | | ------------------------ | -------- | ------ | --------------------- | -| 'diegosouzapw/omniroute' | 'en son' | ~250MB | En son kararlı sürüm | -| 'diegosouzapw/omniroute' | '1.0.3' | ~250MB | Güncel sürüm |--- +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | + +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**YENİ!**OmniRoute artık Windows, macOS ve Linux için**yerel masaüstü uygulaması**olarak mevcut. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -OmniRoute'u bağımsız bir masaüstü uygulaması olarak çalıştırın; yerel modeller için terminal yok, tarayıcı yok, internet gerekmiyor. Elektron tabanlı uygulama şunları içerir: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Yerel Pencere**— Sistem tepsisi entegrasyonuna sahip özel uygulama penceresi -- 🔄**Otomatik Başlat**— Sisteme giriş yapıldığında OmniRoute'u başlatın -- 🔔**Yerel Bildirimler**— Kota tükenmesi veya sağlayıcı sorunlarıyla ilgili uyarılar alın -- ⚡**Tek Tıklamayla Kurulum**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Çevrimdışı Mod**— Birlikte verilen sunucuyla tamamen çevrimdışı çalışır### Hızlı Başlangıç +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Hızlı Başlangıç ```bash # Development mode @@ -979,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -OmniRoute simge durumuna küçültüldüğünde hızlı eylemlerle sistem tepsinizde yaşar: +When minimized, OmniRoute lives in your system tray with quick actions: -- Kontrol panelini aç -- Sunucu bağlantı noktasını değiştirin -- Uygulamadan çık +- Open dashboard +- Change server port +- Quit application -📖 Tüm belgeler: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Seviye | Sağlayıcı | Maliyet | Kota Sıfırlama | En İyisi | -| ------------------ | ------------------------------------ | ----------------------------------------------------- | ------------------ | -------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 ABONELİK** | Claude Kodu (Pro) | 20$/ay | 5 saat + haftalık | Zaten abone oldum | -| | Kodeks (Artı/Pro) | 20-200$/ay | 5 saat + haftalık | OpenAI kullanıcıları | -| | İkizler CLI | **ÜCRETSİZ** | 180K/ay + 1K/gün | Herkes! | -| | GitHub Yardımcı Pilotu | 10-19$/ay | Aylık | GitHub kullanıcıları | -| **🔑API ANAHTARI** | NVIDIA NIM | **ÜCRETSİZ**(sonsuza kadar geliştirme) | ~40 RPM | 70'den fazla açık model | -| | Beyinler | **ÜCRETSİZ**(1 milyon tok/gün) | 60K TPM / 30 RPM | Dünyanın en hızlısı | -| | Büyük | **ÜCRETSİZ**(30 RPM) | 14,4K RPD | Ultra hızlı Lama/Gemma | -| | DeepSeek V3.2 | 1 milyon dolar başına 0,27 dolar/1,10 dolar | Yok | En iyi fiyat/kalite muhakemesi | -| | xAI Grok-4 Hızlı | **1 milyon başına 0,20 ABD Doları/0,50 ABD Doları**🆕 | Yok | En hızlı + araç çağırma, ultra düşük | -| | xAI Grok-4 (standart) | 1 milyon dolar başına 0,20 dolar/1,50 dolar 🆕 | Yok | xAI'den akıl yürütme amiral gemisi | -| | Mistral | Ücretsiz deneme + ücretli | Hız sınırlı | Avrupa Yapay Zekası | -| | OpenRouter | Kullanım başına ödeme | Yok | 100'den fazla model toplamı | -| **💰 UCUZ** | GLM-5 (Z.AI aracılığıyla) 🆕 | 0,5$/1 milyon $ | Günlük 10:00 | 128K çıkış, en yeni amiral gemisi | -| | GLM-4.7 | 0,6 $/1 milyon $ | Günlük 10:00 | Bütçe yedekleme | -| | MiniMax M2.5 🆕 | 0,3 ABD Doları/1 milyon giriş | 5 saatlik ilerleme | Muhakeme + aracılı görevler | -| | MiniMax M2.1 | 0,2$/1 milyon $ | 5 saatlik ilerleme | En ucuz seçenek | -| | Kimi K2.5 (Moonshot API) 🆕 | Kullanım başına ödeme | Yok | Doğrudan Moonshot API erişimi | -| | Kimi K2 | 9$/ay düz | 10 milyon token/ay | Tahmin edilebilir maliyet | -| **🆓 ÜCRETSİZ** | Kod | **$0** | Sınırsız | 5 model sınırsız | -| | Qwen | **$0** | Sınırsız | 4 model sınırsız | -| | Kiro | **$0** | Sınırsız | Claude Sonnet/Haiku (AWS Oluşturucusu) | -| | LongCat Flash-Lite 🆕 | **0$**(50 milyon tok/gün 🔥) | 1RPS | Dünyanın en büyük ücretsiz kotası | -| | Tozlaşma AI 🆕 | **$0**(anahtara gerek yok) | 1 talep/15s | GPT-5, Claude, DeepSeek, Lama 4 | -| | Cloudflare Çalışanları Yapay Zeka 🆕 | **0$**(10.000 Nöron/gün) | ~150 solunum/gün | 50'den fazla model, global üstünlük | -| | Scaleway AI 🆕 | **0$**(toplam 1 milyon jeton) | Hız sınırlı | AB/GDPR, Qwen3 235B, Llama 70B | > 🆕**Yeni modeller eklendi (Mart 2026):**0,20$/0,50$/M fiyatla Grok-4 Fast ailesi (1143 ms ile karşılaştırıldı — Gemini 2.5 Flash'tan %30 daha hızlı), 128K çıkışlı Z.AI aracılığıyla GLM-5, MiniMax M2.5 akıl yürütme, DeepSeek V3.2 güncellenmiş fiyatlandırma, Moonshot doğrudan API aracılığıyla Kimi K2.5. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 0 ABD Doları Kombo Yığın — Tam Ücretsiz Kurulum:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Sıfır maliyet. Kodlamayı asla durdurmayın.**Bunu tek bir OmniRoute birleşimi olarak yapılandırdığınızda tüm geri dönüşler otomatik olarak gerçekleşir; hiçbir zaman manuel geçiş yapılmaz.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Aşağıdaki tüm modeller**sıfır kredi kartı gerekliliğiyle %100 ücretsizdir**. OmniRoute, bir kota dolduğunda bunlar arasında otomatik yönlendirme yapar; kırılamaz 0 ABD doları değerinde bir kombinasyon için hepsini birleştirir.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Modeli | Önek | Sınırı | Oran Limiti | +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | --------------------- | -| `claude-sonnet-4.5` | 'kr/' |**Sınırsız**| Günlük sınır bildirilmedi | -| `Claude-haiku-4.5` | 'kr/' |**Sınırsız**| Günlük sınır bildirilmedi | -| `Claude-opus-4.6` | 'kr/' |**Sınırsız**| Kiro aracılığıyla en son Opus |### 🟢 QODER MODELS (Free PAT via qodercli) +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -| Modeli | Önek | Sınırı | Oran Limiti | +### 🟢 QODER MODELS (Free PAT via qodercli) + +| Model | Prefix | Limit | Rate Limit | | ------------------ | ------ | ------------- | --------------- | -| `kimi-k2-düşünme' | 'eğer/' |**Sınırsız**| Bildirilen sınır yok | -| `qwen3-kodlayıcı-artı` | 'eğer/' |**Sınırsız**| Bildirilen sınır yok | -| 'derin arama-r1' | 'eğer/' |**Sınırsız**| Bildirilen sınır yok | -| 'minimax-m2.1' | 'eğer/' |**Sınırsız**| Bildirilen sınır yok | -| 'kimi-k2' | 'eğer/' |**Sınırsız**| Bildirilen sınır yok | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -> Önerilen bağlantı yöntemi:**Kişisel Erişim Simgesi + `qodercli`**. Tarayıcı OAuth'u (şimdiki değeri) -> `QODER_OAUTH_*` ortam değişkenleri yapılandırılmadığı sürece deneyseldir ve varsayılan olarak devre dışıdır.### 🟡 QWEN MODELS (Device Code Auth) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| Modeli | Önek | Sınırı | Oran Limiti | +### 🟡 QWEN MODELS (Device Code Auth) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | ------------------- | -| `qwen3-kodlayıcı-artı` | 'qw/' |**Sınırsız**| Bildirilen sınır yok | -| `qwen3-kodlayıcı-flaş' | 'qw/' |**Sınırsız**| Bildirilen sınır yok | -| `qwen3-kodlayıcı-sonraki` | 'qw/' |**Sınırsız**| Bildirilen sınır yok | -| 'vizyon modeli' | 'qw/' |**Sınırsız**| Multimodal (resimler) |### 🟣 GEMINI CLI (Google OAuth) +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Modeli | Önek | Sınırı | Oran Limiti | -| ------------------------ | ------ | ---------------------------- | ------------- | -| 'ikizler-3-flash-önizleme' | 'gc/' |**180 bin tok/ay**+ 1 bin/gün | Aylık sıfırlama | -| 'ikizler-2.5-pro' | 'gc/' | 180K/ay (ortak havuz) | Yüksek kalite |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +### 🟣 GEMINI CLI (Google OAuth) -| Seviye | Günlük Limit | Oran Limiti | Notlar | -| ---------- | ------------ | ----------- | -------------------------------------------- | -| Ücretsiz (Geliştirme) | Jeton sınırı yok |**~40 RPM**| 70+ model; 2025 ortalarında saf faiz limitlerine geçiş | +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -Popüler ücretsiz modeller: "moonshotai/kimi-k2.5" (Kimi K2.5), "z-ai/glm4.7" (GLM 4.7), "deepseek-ai/deepseek-v3.2" (DeepSeek V3.2), "nvidia/llama-3.3-70b-instruct", "deepseek/deepseek-r1"### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) -| Seviye | Günlük Limit | Oran Limiti | Notlar | -| ---- | ----------------- | ---------------- | --------------------------------- | -| Ücretsiz |**1 milyon token/gün**| 60K TPM / 30 RPM | Dünyanın en hızlı Yüksek Lisans çıkarımı; günlük olarak sıfırlanır | +| Tier | Daily Limit | Rate Limit | Notes | +| ---------- | ------------ | ----------- | ------------------------------------------------------ | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -Ücretsiz olarak mevcuttur: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -| Seviye | Günlük Limit | Oran Limiti | Notlar | +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) + +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | + +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` + +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---- | ------------- | ---------------- | ----------------------------------------- | -| Ücretsiz |**14,4K RPD**| Model başına 30 RPM | Kredi kartı yok; 429 limitli, ücret alınmadı | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -Ücretsiz olarak mevcuttur: "llama-3.3-70b-versatile", "gemma2-9b-it", "mixtral-8x7b", "whisper-large-v3"### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -| Modeli | Önek | Günlük Ücretsiz Kota | Notlar | +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 + +| Model | Prefix | Daily Free Quota | Notes | | ----------------------------- | ------ | ----------------- | ----------------------- | -| 'LongCat-Flash-Lite' | 'lc/' |**50 milyon jeton**💥 | Şimdiye kadarki en büyük ücretsiz kota | -| 'LongCat-Flash-Sohbet' | 'lc/' | 500.000 jeton | Çok turlu sohbet | -| 'LongCat-Flash-Düşünme' | 'lc/' | 500.000 jeton | Muhakeme / CoT | -| 'LongCat-Flash-Düşünme-2601' | 'lc/' | 500.000 jeton | Ocak 2026 versiyonu | -| 'LongCat-Flash-Omni-2603' | 'lc/' | 500.000 jeton | Çok modlu | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | -> Herkese açık beta sürümünde %100 ücretsiz. [longcat.chat](https://longcat.chat) adresinden e-posta veya telefonla kaydolun. Günlük 00:00 UTC'yi sıfırlar.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. -| Modeli | Önek | Oran Limiti | Sağlayıcı Arkasında | +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| 'açık' | 'pol/' | 1 talep/15s | GPT-5 | -| 'Claude' | 'pol/' | 1 talep/15s | Antropik Claude | -| 'İkizler' | 'pol/' | 1 talep/15s | Google İkizler | -| 'derin arama' | 'pol/' | 1 talep/15s | DeepSeek V3 | -| 'lama' | 'pol/' | 1 talep/15s | Meta Lama 4 İzci | -| 'mistral' | 'pol/' | 1 talep/15s | Mistral AI | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**Sıfır sürtünme:**Kayıt yok, API anahtarı yok. Tozlaşma sağlayıcısını boş bir anahtar alanıyla ekleyin ve hemen çalışır.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| Seviye | Günlük Nöronlar | Eşdeğer Kullanım | Notlar | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 + +| Tier | Daily Neurons | Equivalent Usage | Notes | | ---- | ------------- | --------------------------------------- | ----------------------- | -| Ücretsiz |**10.000**| ~150 Yüksek Lisans / 500s ses / 15K yerleştirme | Global uç, 50'den fazla model | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -Popüler ücretsiz modeller: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (ücretsiz ses!), `@cf/qwen/qwen2.5-coder-15b-instruct` +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -> [dash.cloudflare.com](https://dash.cloudflare.com) adresinden API Jetonu + Hesap Kimliği gerektirir. Hesap Kimliğini sağlayıcı ayarlarında saklayın.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -| Seviye | Ücretsiz Kota | Konum | Notlar | +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | | ---- | ------------- | ------------ | ----------------------------------- | -| Ücretsiz |**1 milyon jeton**| 🇫🇷 Paris, AB | Limitler dahilinde kredi kartına gerek yok | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | -Ücretsiz olarak mevcuttur: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` -> AB/GDPR ile uyumludur. [console.scaleway.com](https://console.scaleway.com) adresinden API anahtarını alın. +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). ->**💡 Ultimate Ücretsiz Yığın (11 Sağlayıcı, Sonsuza Kadar 0 ABD Doları):** +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Sonnet/Haiku SINIRSIZ -> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 SINIRSIZ -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 milyon jeton/gün 🔥 -> Tozlaşma (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — anahtara gerek yok -> Qwen (qw/) → qwen3 kodlayıcı modelleri SINIRSIZ -> Gemini (gemini/) → Gemini 2.5 Flash — 1.500 talep/gün ücretsiz -> Cloudflare AI (cf/) → 50'den fazla model — 10.000 Nöron/gün -> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1 milyon ücretsiz token (AB) -> Groq (groq/) → Llama/Gemma — 14,4K talep/gün ultra hızlı -> NVIDIA NIM (nvidia/) → 70'ten fazla açık model — sonsuza kadar 40 RPM -> Cerebras (cerebras/) → Lama/Qwen dünyanın en hızlısı — 1 milyon tok/gün -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Herhangi bir sesi/videoyu**0$**karşılığında yazıya dökün — Deepgram 200$ ücretsiz, AssemblyAI 50$ geri dönüş ve sınırsız acil durum yedeklemesi olarak Groq Whisper ile önde gidiyor. +## 🎙️ Free Transcription Combo -| Sağlayıcı | Ücretsiz Krediler | En İyi Model | Oran Limiti | -| ----------------- | ----------------------- | --------------------------------- | ---------------------------- | -| 🟢**Deepgram**|**200$ ücretsiz**(kayıt) | `nova-3` — en iyi doğruluk, 30'dan fazla dilde | Ücretsiz kredilerde RPM sınırı yok | -| 🔵**AssemblyAI**|**50$ ücretsiz**(kayıt) | `universal-3-pro` — bölümler, görüş, PII | Ücretsiz kredilerde RPM sınırı yok | -| 🔴**Groq**|**Sonsuza kadar ücretsiz**| "whisper-large-v3" — OpenAI Whisper | 30 RPM (hız sınırlı) | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. -**`/dashboard/combos`ta önerilen kombinasyon:**``` +| Provider | Free Credits | Best Model | Rate Limit | +| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | + +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Daha sonra `/dashboard/media` →**Transkripsiyon**sekmesinde: herhangi bir ses veya video dosyasını yükleyin → birleşik uç noktanızı seçin → desteklenen formatlarda transkripsiyon alın.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 yalnızca geçiş proxy'si değil, operasyonel bir platform olarak oluşturulmuştur.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Özellik | Ne İşe Yarar | -| -------------------------------------- | --------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Grok-4 Hızlı Ailesi** | xAI modelleri 0,20 ABD Doları/0,50 ABD Doları/M seviyesinde — karşılaştırmalı 1143 ms (Gemini 2.5 Flash'tan %30 daha hızlı) | -| 🧠**Z.AI aracılığıyla GLM-5** | 128K çıktı bağlamı, 0,5 Milyon Dolar/1 Milyon Dolar — GLM ailesinin en yeni amiral gemisi | -| 🔮**MiniMax M2.5** | Muhakeme + aracılı görevler 0,30 ABD Doları/1 Milyon Dolar — M2.1'e göre önemli bir artış | -| 🎯**Model başına toolCalling Flag** | Kayıt defterinde her model için 'toolCalling: doğru/yanlış' — AutoCombo, araç özelliği olmayan modelleri atlar | -| 🌍**Çok Dilli Niyet Tespiti** | AutoCombo puanlamada PT/ZH/ES/AR anahtar sözcükleri — İngilizce olmayan içerik için daha iyi model seçimi | -| 📊**Kıyaslama Odaklı Geri Dönüşler** | Canlı isteklerden elde edilen gerçek p95 gecikmesi birleşik puanlamayı besler — AutoCombo gerçek verilerden öğrenir | -| 🔁**Tekilleştirme Talebi** | İçerik karması tabanlı yinelenenleri kaldırma penceresi — çoklu aracı açısından güvenli, mükerrer ödemeleri önler | -| 🔌**Takılabilir YönlendiriciStrateji** | Genişletilebilir `RouterStrategy` arayüzü — eklenti olarak özel yönlendirme mantığı ekleyin | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Özellik | Ne İşe Yarar | -| ------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Model Oyun Alanı** | Herhangi bir modeli doğrudan test etmek için kontrol paneli sayfası — sağlayıcı/model/uç nokta seçiciler, Monaco Düzenleyici, akış, iptal, zamanlama | -| 🔏**CLI Parmak İzi Eşleştirme** | Yerel CLI imzalarıyla eşleşmek için sağlayıcı başına başlık/gövde sıralaması - Ayarlar > Güvenlik'te sağlayıcıya göre geçiş yapın.**Proxy IP'niz korunur** | -| 🤝**ACP Desteği (Ajan İstemci Protokolü)** | CLI aracısı keşfi (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 tane daha), süreç oluşturucu, `/api/acp/agents' uç noktası | -| 🤖**ACP Temsilcileri Kontrol Paneli** | Hata ayıklama › Aracılar sayfası — herhangi bir CLI aracı için yükleme durumu, sürüm ve özel aracı formunu içeren 14 aracıdan oluşan tablo.**OpenCode**kullanıcıları, mevcut tüm modellerle birlikte kullanıma hazır bir yapılandırmayı otomatik olarak oluşturan bir "opencode.json'u indir" düğmesine sahip olur. | -| 🔧**Özel Model `apiFormat` Yönlendirme** | `apiFormat: "responses"` içeren özel modeller artık Responses API çeviricisine doğru şekilde yönlendiriliyor | -| 🏢**Codex Çalışma Alanı Yalıtımı** | E-posta başına birden fazla Codex çalışma alanı — OAuth, bağlantıları çalışma alanı kimliğine göre doğru şekilde ayırır | -| 🔄**Elektron Otomatik Güncelleme** | Masaüstü uygulaması güncellemeleri kontrol eder + yeniden başlatıldığında otomatik yükleme | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Özellik | Ne İşe Yarar | -| -------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**MCP Sunucusu (25 araç)** | 3 aktarım yoluyla IDE/aracı araçları: stdio, SSE (`/api/mcp/sse`), Akış Yapılabilir HTTP (`/api/mcp/stream`). 18 çekirdek + 3 bellek + 4 beceri aracı | -| 🤝**A2A Sunucusu (JSON-RPC + SSE)** | Senkronizasyon ve akış akışlarıyla aracıdan aracıya görev yürütme | -| 🧭**Birleştirilmiş Uç Noktalar Sayfası** | Uç Nokta Proxy'si, MCP, A2A ve API Uç Noktaları sekmelerini içeren sekmeli yönetim sayfası | -| 🎚️**Hizmeti Etkinleştirme/Devre Dışı Bırakma Geçişleri** | MCP ve A2A için ayarların kalıcı olduğu AÇIK/KAPALI anahtarlar (varsayılan: KAPALI) | -| 🛰️**MCP Çalışma Zamanı Kalp Atışı** | Gerçek süreç durumu (pid, çalışma süresi, kalp atışı yaşı, aktarım, kapsam modu) | -| 📋**MCP Denetim Kaydı** | Başarı/başarısızlık ve anahtar ilişkilendirme içeren filtrelenebilir denetim günlükleri | -| 🔐**MCP Kapsamının Uygulanması** | Kontrollü araç erişimi için 10 ayrıntılı kapsam izni | -| 📡**A2A Görev Yaşam Döngüsü Yönetimi** | Görevleri listeleyin/filtreleyin, olayları/yapıları inceleyin, çalışan görevleri iptal edin | -| 📋**Acente Kartı Keşfi** | İstemcinin otomatik keşfi için `/.well-known/agent.json` | -| 🧪**Protokol E2E Test Donanımı** | Gerçek MCP SDK + A2A istemcisi `test:protokoller:e2e`de akar | -| ⚙️**Operasyonel Kontroller** | Tek bir kontrol yüzeyinden komboyu değiştirin, dayanıklılık profilleri uygulayın, kesicileri sıfırlayın | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Özellik | Ne İşe Yarar | -| --------------------------------------- | ----------------------------------------------------------------------------------- | ----------------------- | -| 🎯**Akıllı 4 Katmanlı Geri Dönüş** | Otomatik yönlendirme: Abonelik → API Anahtarı → Ucuz → Ücretsiz | -| 📊**Gerçek Zamanlı Kota Takibi** | Canlı token sayısı + sağlayıcı başına geri sayımı sıfırlayın | -| 🔄**Biçim Çevirisi** | OpenAI ↔ Claude ↔ Gemini ↔ Şema güvenli dönüşümlerle yanıtlar | -| 👥**Çoklu Hesap Desteği** | Akıllı seçimle sağlayıcı başına birden fazla hesap | -| 🔄**Otomatik Jeton Yenileme** | OAuth belirteçleri yeniden denemeyle otomatik olarak yenilenir | -| 🎨**Özel Kombinasyonlar** | 9 dengeleme stratejisi + geri dönüş zinciri kontrolü | -| 🌐**Joker Karakter Yönlendirici** | 'sağlayıcı/\*' dinamik yönlendirme | -| 🧠**Bütçe Kontrollerini Düşünmek** | Geçiş, otomatik, özel ve uyarlanabilir akıl yürütme sınırları | -| 🔀**Model Takma Adları** | Yerleşik + özel model örtüşme ve geçiş güvenliği | -| ⚡**Arka Plan Bozulması** | Düşük öncelikli arka plan görevlerini daha ucuz modellere yönlendirin | -| 🧪**Göreve Duyarlı Akıllı Yönlendirme** | Modeli içerik türüne göre otomatik seç (kodlama/vizyon/analiz/özetleme) | -| 🔄**A2A Temsilci İş Akışları** | Durum bilgisi olan çok adımlı aracı yürütmeleri için deterministik FSM orkestratörü | -| 🔀**Uyarlanabilir Yönlendirme** | Belirteç hacmine ve istem karmaşıklığına dayalı dinamik strateji geçersiz kılma | -| 🎲**Sağlayıcı Çeşitliliği** | Shannon entropi puanlaması otomatik birleşik trafik dağıtımını dengeleme | -| 💬**Sistem İstemi Ekleme** | Tutarlı bir şekilde uygulanan küresel davranış kontrolleri | -| 📄**Yanıtlar API Uyumluluğu** | Codex ve gelişmiş aracılı iş akışları için tam `/v1/responses` desteği | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Özellik | Ne İşe Yarar | -| ------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ---------------------------------------- | -| 🖼️**Görüntü Oluşturma** | Bulut ve yerel arka uçlarla `/v1/images/ Generations` | -| 📐**Gömmeler** | arama ve RAG ardışık düzenleri için `/v1/embeddings` | -| 🎤**Ses Transkripsiyonu** | `/v1/audio/transcriptions` — 7 sağlayıcı (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), otomatik dil algılama, MP4/MP3/WAV desteği | -| 🔊**Metinden Konuşmaya** | `/v1/audio/speech` — Doğru hata mesajlarına sahip 10 sağlayıcı (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) | -| 🎬**Video Oluşturma** | `/v1/videos/ Generations` (ComfyUI + SD WebUI iş akışları) | -| 🎵**Müzik Üretimi** | `/v1/music/jenerasyonlar` (ComfyUI iş akışları) | -| 🛡️**Denetimler** | `/v1/moderations` güvenlik kontrolleri | -| 🔀**Yeniden sıralama** | alaka puanlaması için `/v1/rerank` | -| 🔍**Web Araması**🆕 | `/v1/search` — 5 sağlayıcı (Serper, Brave, Perplexity, Exa, Tavily), 6.500'den fazla ücretsiz/ay, otomatik yük devretme, önbellek | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Özellik | Ne İşe Yarar | -| ------------------------------------------------- | ---------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**Devre Kesiciler** | Eşik kontrolleriyle model başına açma/kurtarma | -| 🎯**Uç Nokta Farkındalığı Olan Modeller** | Özel modeller desteklenen uç noktaları + API formatını bildirir | -| 🛡️**Yıldırım Karşıtı Sürü** | Yeniden deneme/oranlandırma olaylarında Mutex + semafor korumaları | -| 🧠**Anlamsal + İmza Önbelleği** | İki önbellek katmanıyla maliyet/gecikme azalması | -| ⚡**İdempotentlik Talebi** | Yinelenen koruma penceresi | -| 🔒**TLS Parmak İzi Sahtekarlığı** | Tarayıcı benzeri TLS parmak izi —**bot tespitini ve hesap işaretlemeyi azaltır** | -| 🔏**CLI Parmak İzi Eşleştirme** | Yerel CLI istek imzalarıyla eşleşir —**proxy IP'sini korurken yasaklama riskini azaltır** | -| 🌐**IP Filtreleme** | Açığa çıkan dağıtımlar için izin verilenler listesi/engellenenler listesi kontrolü | -| 📊**Düzenlenebilir Oran Limitleri** | Kalıcılıkla yapılandırılabilir küresel/sağlayıcı düzeyinde sınırlar | -| 📉**Zarif Bozulma** | Çekirdek ağ geçidi işlemlerini koruyan çok katmanlı yetenek geri dönüşleri | -| 📜**Yapılandırma Denetim İzi** | Basit geri alma işlemleriyle operasyonel sapmayı önleyen fark tabanlı değişiklik takibi | -| ⏳**Sağlayıcı Sağlık Senkronizasyonu** | Yetkilendirme hatalarından önce uyarıları tetikleyen proaktif belirteç süre sonu izleme | -| 🚪**Yasaklı Hesapları Otomatik Devre Dışı Bırak** | Operasyonel devre kesici, kalıcı olarak engellenen token hesaplarını otomatik olarak kapatıyor | -| 🔑**API Anahtar Yönetimi + Kapsam Belirleme** | Güvenli anahtar verme/döndürme ve model/sağlayıcı kontrolleri | -| 👁️**Kapsamlı API Anahtarı Gösterimi**🆕 | API anahtarlarının "ALLOW_API_KEY_REVEAL" aracılığıyla kurtarılmasını etkinleştirme | -| 🛡️**Korumalı `/modeller`** | Model kataloğu için isteğe bağlı kimlik doğrulama ve sağlayıcı gizleme | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Özellik | Ne İşe Yarar | -| --------------------------------------- | -------------------------------------------------------------------------------- | ---------------------------- | -| 📝**İstek + Proxy Günlüğü** | Tam istek/yanıt ve proxy günlüğü | -| 📉**Akışlı Ayrıntılı Günlükler**🆕 | SSE verisi akışlarını kullanıcı arayüzüne temiz bir şekilde yeniden yapılandırır | -| 📋**Birleşik Günlükler Kontrol Paneli** | Tek sayfada istek, proxy, denetim ve konsol görünümleri | -| 🔍**Telemetri İste** | p50/p95/p99 gecikmesi ve istek takibi | -| 🏥**Sağlık Kontrol Paneli** | Çalışma süresi, kesici durumları, kilitlenmeler, önbellek istatistikleri | -| 💰**Maliyet Takibi** | Bütçe kontrolleri ve model başına fiyatlandırma görünürlüğü | -| 📈**Analitik Görselleştirmeler** | Model/sağlayıcı kullanım bilgileri ve trend görünümleri | -| 🧪**Değerlendirme Çerçevesi** | Yapılandırılabilir maç stratejileriyle altın set testi | -| 📡**Canlı Teşhis**🆕 | Doğru birleşik canlı test için semantik önbellek atlama | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Özellik | Ne İşe Yarar | -| ----------------------------------- | --------------------------------------------------------------------------------- | -------------------------------------- | -| 🌐**Her Yere Dağıtın** | Localhost, VPS, Docker, Bulut ortamları | -| 🚇**Cloudflare Tüneli**🆕 | Kontrol panelinden tek tıklamayla Hızlı Tünel entegrasyonu | -| 🔑**API Anahtar Modeli Filtreleme** | Yerel /v1/models yanıtı, atanmış Taşıyıcı bağlam rolleri aracılığıyla filtrelendi | -| ⚡**Akıllı Önbellek Atlaması** | Yapılandırılabilir TTL buluşsal yöntemi ve zorunlu yeniden getirme kontrolleri | -| 🔄**Yedekle/Geri Yükle** | İhracat/ithalat ve olağanüstü durum kurtarma akışları | -| 🧙**İlk Katılım Sihirbazı** | İlk çalıştırmada kılavuzlu kurulum | -| 🔧**CLI Araçları Kontrol Paneli** | Popüler kodlama araçları için tek tıkla kurulum | -| 🎮**Model Oyun Alanı** | Kontrol panelinden herhangi bir sağlayıcıyı/modeli/uç noktayı test edin | -| 🔏**CLI Parmak İzi Geçişi** | Ayarlar > Güvenlik | Sağlayıcıya göre parmak izi eşleştirme | -| 🌐**i18n (30 dil)** | Tam kontrol paneli + RTL kapsamıyla belge dili desteği | -| 🧹**Tüm Modelleri Temizle** | Sağlayıcı ayrıntılarında tek tıklamayla model listesi temizleme | -| 👁️**Kenar Çubuğu Kontrolleri**🆕 | Görünüm Ayarları'ndan bileşenleri ve entegrasyonları gizleyin | -| 📋**Sayı Şablonları** | Hatalar ve özellikler için standartlaştırılmış GitHub şablonları | -| 📂**Özel Veri Dizini** | Depolama konumu için `DATA_DIR` geçersiz kılma | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1292,103 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -Kota, oran veya sağlık başarısız olduğunda OmniRoute, manuel geçişe gerek kalmadan otomatik olarak bir sonraki adaya geçer.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A, kullanıcı arayüzünde ve belgelerde keşfedilebilir (gizli değil) -- Protokol durumu API'leri canlı operasyonel verileri açığa çıkarır (`/api/mcp/*`, `/api/a2a/*`) -- Kontrol panelleri 2. gün operasyonlarına yönelik eylemleri içerir (kombo geçişler, devre kesici sıfırlamaları, görev iptali)#### Translator + validation workflow +#### Protocol management that is visible and operable -Çevirmen alanı şunları içerir: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Oyun Alanı**: dönüşüm kontrolleri isteyin -**Sohbet Test Cihazı**: tam istek/yanıt gidiş-dönüş -**Test Tezgahı**: tek çalıştırmada birden fazla vaka -**Canlı Monitör**: gerçek zamanlı trafik görünümü +#### Translator + validation workflow -Ayrıca "npm çalıştırma testi:protokoller:e2e" aracılığıyla gerçek istemcilerle protokol doğrulama. +The Translator area includes: -> 📖**[MCP Sunucusu README](open-sse/mcp-server/README.md)**— Araç referansı, IDE yapılandırmaları ve istemci örnekleri +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A Sunucusu README](src/lib/a2a/README.md)**— Beceriler, JSON-RPC yöntemleri, akış ve görev yaşam döngüsü## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute, LLM yanıt kalitesini altın bir sete göre test etmek için yerleşik bir değerlendirme çerçevesi içerir. Kontrol panelindeki**Analytics → Evals**aracılığıyla bu bilgilere erişin.### Built-in Golden Set +## 🧪 Evaluations (Evals) -Önceden yüklenmiş "OmniRoute Golden Set" aşağıdakiler için test senaryoları içerir: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Selamlar, matematik, coğrafya, kod oluşturma -- JSON formatı uyumluluğu, çeviri, indirim oluşturma -- Güvenlik reddi (zararlı içerik), sayma, boole mantığı### Evaluation Strategies +### Built-in Golden Set -| Strateji | Açıklama | Örnek | -| -------------- | ------------------------------------------------------------ | ------------------------------- | --- | -| 'kesin' | Çıktı tam olarak eşleşmelidir | `"4"` | -| 'içerir' | Çıktı alt dize içermelidir (büyük/küçük harfe duyarlı değil) | ''Paris'' | -| 'normal ifade' | Çıktı normal ifade düzeniyle eşleşmelidir | `"1.*2.*3"` | -| 'özel' | Özel JS işlevi doğru/yanlış değerini döndürür | `(çıkış) => çıktı.uzunluk > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) - -🧩 MCP Kurulumu (Model Bağlam Protokolü) +
+🧩 MCP Setup (Model Context Protocol) -MCP aktarımını stdio modunda başlatın:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Önerilen doğrulama akışı: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. MCP istemcinizi stdio üzerinden bağlayın. -2. 'omniroute_get_health'i çalıştırın. -3. 'omniroute_list_combos'u çalıştırın. -4. Kalp atışını, aktiviteyi ve denetimi onaylamak için `/dashboard/mcp`yi açın. +Useful APIs for automation: -Otomasyon için faydalı API'ler: +- `GET /api/mcp/status` +- `GET /api/mcp/tools` +- `GET /api/mcp/audit` +- `GET /api/mcp/audit/stats` -- '/api/mcp/status'u AL' -- '/api/mcp/tools'U ALIN' -- '/api/mcp/audit'i AL' -- '/api/mcp/audit/stats'ı AL'
+ - -🤝 A2A Kurulumu (Agent2Agent) +
+🤝 A2A Setup (Agent2Agent) -Temsilciyi keşfedin:```bash +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Bir görev gönder:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` +Manage lifecycle: -Yaşam döngüsünü yönetin: +- `GET /api/a2a/status` +- `GET /api/a2a/tasks` +- `GET /api/a2a/tasks/:id` +- `POST /api/a2a/tasks/:id/cancel` -- '/api/a2a/status'u AL' -- '/api/a2a/tasks'ı AL' -- `AL /api/a2a/tasks/:id' -- 'POST /api/a2a/tasks/:id/iptal' +Operational UI: -Operasyonel kullanıcı arayüzü: +- `/dashboard/a2a` for task/state/stream observability and smoke actions -- Görev/durum/akış gözlemlenebilirliği ve duman eylemleri için `/dashboard/a2a`
+ - -🧪 Uçtan uca protokol doğrulama +
+🧪 End-to-end protocol validation -Her iki protokolü de gerçek istemcilerle doğrulayın:```bash +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Bu şunları doğrular: +This verifies: -- MCP SDK istemcisi bağlantısı/listesi/çağrısı -- A2A keşfi/gönderme/akış/alma/iptal etme -- MCP denetimi ve A2A görev yönetimi API'lerindeki verileri çapraz kontrol edin
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs - -💳 Abonelik Sağlayıcıları### Claude Code (Pro/Max) + + +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1401,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Profesyonel İpucu:**Karmaşık görevler için Opus'u, hız için Sonnet'i kullanın. OmniRoute model başına kotayı takip eder!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1415,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Artık her Codex hesabının 'Kontrol Paneli -> Sağlayıcılar'da politika geçişleri var: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5sa' (AÇIK/KAPALI): 5 saatlik pencere eşik politikasını uygulayın. -- `Haftalık' (AÇIK/KAPALI): haftalık pencere eşik politikasını uygulayın. -- Eşik davranışı: etkin bir pencere >=%90 kullanıma ulaştığında, o hesap atlanır. -- Döndürme davranışı: OmniRoute bir sonraki uygun Codex hesabına otomatik olarak yönlendirir. -- Sıfırlama davranışı: sağlayıcının 'resetAt' süresi geçtiğinde hesap otomatik olarak tekrar uygun hale gelir. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Senaryolar: +Scenarios: -- "5h ON" + "Weekly ON": pencerelerden herhangi biri eşiğe ulaştığında hesap atlanır. -- "5 saat KAPALI" + "Haftalık AÇIK": yalnızca haftalık kullanım hesabı engelleyebilir. -- "5 saat AÇIK" + "Haftalık KAPALI": yalnızca 5 saatlik kullanım hesabı bloke edebilir. -- `resetAt' başarılı oldu: hesap otomatik olarak rotasyona yeniden giriyor (manuel yeniden etkinleştirme yok).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1440,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**En İyi Değer:**Devasa ücretsiz katman! Bunu ücretli katmanlardan önce kullanın.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1455,71 +1662,91 @@ Models:
- -🔑 API Anahtarı Sağlayıcıları### NVIDIA NIM (FREE developer access — 70+ models) +
+🔑 API Key Providers + +### NVIDIA NIM (FREE developer access — 70+ models) 1. Sign up: [build.nvidia.com](https://build.nvidia.com) -2. Ücretsiz API anahtarını edinin (1000 çıkarım kredisi dahil) -3. Kontrol Paneli → Sağlayıcı Ekle → NVIDIA NIM: - - API Anahtarı: `nvapi-anahtarınız` +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Modeller:**"nvidia/llama-3.3-70b-instruct", "nvidia/mistral-7b-instruct" ve 50'den fazla model +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -**Profesyonel İpucu:**OpenAI uyumlu API — OmniRoute'un format çevirisiyle sorunsuz şekilde çalışır!### DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -1. Kaydolun: [platform.deepseek.com](https://platform.deepseek.com) -2. API anahtarını alın -3. Kontrol Paneli → Sağlayıcı Ekle → DeepSeek +### DeepSeek -**Modeller:**"deepseek/deepseek-sohbet", "deepseek/deepseek-kodlayıcı"### Groq (Free Tier Available!) +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -1. Kaydolun: [console.groq.com](https://console.groq.com) -2. API anahtarını alın (ücretsiz katman dahil) -3. Kontrol Paneli → Sağlayıcı Ekle → Groq +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Modeller:**"groq/llama-3.3-70b", "groq/mixtral-8x7b" +### Groq (Free Tier Available!) -**Profesyonel İpucu:**Ultra hızlı çıkarım — gerçek zamanlı kodlama için en iyisi!### OpenRouter (100+ Models) +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -1. Kaydolun: [openrouter.ai](https://openrouter.ai) -2. API anahtarını alın -3. Kontrol Paneli → Sağlayıcı Ekle → OpenRouter +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Modeller:**Tek bir API anahtarı aracılığıyla tüm büyük sağlayıcıların 100'den fazla modeline erişin. +**Pro Tip:** Ultra-fast inference — best for real-time coding! -**Kontrol Paneli davranışı:**OpenRouter modelleri**Kullanılabilir Modeller**'den yönetilir. Manuel ekleme, içe aktarma ve otomatik senkronizasyonun tümü aynı listeyi günceller.
+### OpenRouter (100+ Models) - -💰 Ucuz Sağlayıcılar (Yedekleme)### GLM-4.7 (Daily reset, $0.6/1M) +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -1. Kaydolun: [Zhipu AI](https://open.bigmodel.cn/) -2. Kodlama Planından API anahtarını alın -3. Kontrol Paneli → API Anahtarı Ekle: - - Sağlayıcı: `glm` - - API Anahtarı: "anahtarınız" +**Models:** Access 100+ models from all major providers through a single API key. -**Kullanım:**`glm/glm-4.7` +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -**Profesyonel İpucu:**Kodlama Planı, 1/7 maliyetle 3 kat kota sunuyor! Her gün sabah 10:00'a sıfırlayın.### MiniMax M2.1 (5h reset, $0.20/1M) + -1. Kaydolun: [MiniMax](https://www.minimax.io/) -2. API anahtarını alın -3. Kontrol Paneli → API Anahtarı Ekle +
+💰 Cheap Providers (Backup) -**Kullanım:**`minimax/MiniMax-M2.1` +### GLM-4.7 (Daily reset, $0.6/1M) -**Profesyonel İpucu:**Uzun bağlam için en ucuz seçenek (1 milyon token)!### Kimi K2 ($9/month flat) +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -1. Abone olun: [Moonshot AI](https://platform.moonshot.ai/) -2. API anahtarını alın -3. Kontrol Paneli → API Anahtarı Ekle +**Use:** `glm/glm-4.7` -**Kullanım:**`kimi/kimi-latest` +**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. -**Profesyonel İpucu:**10 milyon token için ayda sabit 9 ABD doları = 0,90 ABD doları/1 milyon etkin maliyet!
+### MiniMax M2.1 (5h reset, $0.20/1M) - -🆓 ÜCRETSİZ Sağlayıcılar (Acil Durum Yedekleme)### Qoder (5 FREE models via OAuth) +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `minimax/MiniMax-M2.1` + +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1560,8 +1787,10 @@ Models:
- -🎨 Kombinasyonlar Oluşturun### Example 1: Maximize Subscription → Cheap Backup +
+🎨 Create Combos + +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1589,8 +1818,10 @@ Cost: $0 forever!
- -🔧 CLI Entegrasyonu### Cursor IDE +
+🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1601,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Tek tıklamayla yapılandırma için kontrol panelindeki**CLI Araçları**sayfasını kullanın veya `~/.claude/settings.json` dosyasını manuel olarak düzenleyin.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1612,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Seçenek 1 — Kontrol Paneli (önerilen):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Seçenek 2 — Manuel:**`~/.openclaw/openclaw.json`ı düzenleyin:```json +```json { "models": { "providers": { @@ -1629,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Not:**OpenClaw yalnızca yerel OmniRoute ile çalışır. IPv6 çözümleme sorunlarını önlemek için "localhost" yerine "127.0.0.1" kullanın.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1643,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**1. Adım:**OmniRoute'u özel sağlayıcı olarak ekleyin:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**2. Adım:**Proje kökünüzde `opencode.json` dosyasını oluşturun/düzenleyin:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1669,118 +1909,130 @@ opencode } } } -```` +``` -**3. Adım:**OpenCode'da modeli seçin:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**İpucu:**OmniRoute `/v1/models` uç noktanızda bulunan herhangi bir modeli `modeller` bölümüne ekleyin. OmniRoute kontrol panelinizden 'sağlayıcı/model kimliği' biçimini kullanın.
+ --- ## Sorun Giderme - -Sorun giderme kılavuzunu genişletmek için tıklayın +
+Click to expand troubleshooting guide -**"Dil modeli mesaj sağlamıyordu"** +**"Language model did not provide messages"** -- Sağlayıcı kotası tükendi → Kontrol paneli kotası izleyicisini kontrol edin -- Çözüm: Kombine geri dönüş kullanın veya daha ucuz seviyeye geçin +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**Hız sınırlaması** +**Rate limiting** -- Abonelik kotası doldu → GLM/MiniMax'a geri dönüş -- Kombinasyon ekleyin: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**OAuth jetonunun süresi doldu** +**OAuth token expired** -- OmniRoute tarafından otomatik olarak yenilendi -- Sorunlar devam ederse: Kontrol Paneli → Sağlayıcı → Yeniden Bağlan +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Yüksek maliyetler** +**High costs** -- Kontrol Paneli → Maliyetler'deki kullanım istatistiklerini kontrol edin -- Birincil modeli GLM/MiniMax'a değiştirin -- Kritik olmayan görevler için ücretsiz katmanı (Gemini CLI, Qoder) kullanın +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Kontrol Paneli/API bağlantı noktaları yanlış** +**Dashboard/API ports are wrong** -- `PORT' standart temel bağlantı noktasıdır (ve varsayılan olarak API bağlantı noktasıdır) -- `API_PORT` yalnızca OpenAI uyumlu API dinleyicisini geçersiz kılar -- "DASHBOARD_PORT" yalnızca kontrol panelini/Next.js dinleyicisini geçersiz kılar -- `NEXT_PUBLIC_BASE_URL`yi kontrol panelinize/genel URL'nize ayarlayın (OAuth geri aramaları için) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Bulut senkronizasyon hataları** +**Cloud sync errors** -- `BASE_URL`nin çalışan örneğinize işaret ettiğini doğrulayın -- "CLOUD_URL"nin beklenen bulut uç noktanıza işaret ettiğini doğrulayın -- `NEXT_PUBLIC_*` değerlerini sunucu tarafı değerleriyle uyumlu tutun +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**İlk giriş çalışmıyor** +**First login not working** -- `.env`de `INITIAL_PASSWORD`u kontrol edin -- Ayarlanmadığında geri dönüş şifresi "123456" olur +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**İstek günlüğü yok** +**No request logs** -- İstek yapıları, istek başına bir JSON dosyası olarak "DATA_DIR/call_logs/" dizinine yazılır -- Aşama başına ayrıntılı yüklere ihtiyacınız varsa Kontrol Paneli → Günlükler → Günlükleri İste'den ardışık düzen yakalamayı etkinleştirin -- Uygulama konsolu günlüklerinin de "logs/application/app.log" içinde olmasını istiyorsanız "APP_LOG_TO_FILE=true" ayarını yapın -- `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` ve `CALL_LOG_MAX_ENTRIES`'i gerektiği gibi ayarlayın +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Bağlantı testi OpenAI uyumlu sağlayıcılar için "Geçersiz" olduğunu gösteriyor** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Birçok sağlayıcı `/models` uç noktasını göstermez +- Many providers don't expose a `/models` endpoint - OmniRoute v1.0.6+ includes fallback validation via chat completions -- Temel URL'nin `/v1' sonekini içerdiğinden emin olun### 🔐 OAuth on a Remote Server +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ OmniRoute'u VPS, Docker veya herhangi bir uzak sunucuda çalıştıran kullanıcılar için önemlidir**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -**Antigravity**ve**Gemini CLI**sağlayıcıları**Google OAuth 2.0**kullanır. Google, OAuth akışındaki "redirect_uri"nin, uygulamanın Google Cloud Console'daki önceden kayıtlı URI'lerden biriyle tam olarak eşleşmesini gerektirir. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -OmniRoute'ta bir araya getirilen OAuth kimlik bilgileri**yalnızca "localhost" için**kaydedilir. OmniRoute'a uzak bir sunucudan (ör. `https://omniroute.myserver.com`) eriştiğinizde Google, kimlik doğrulamayı şu şekilde reddeder:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Google Cloud Console'da sunucunuzun URI'sıyla bir**OAuth 2.0 İstemci Kimliği**oluşturmanız gerekir.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Google Cloud Console'u açın** +#### Step-by-step -Şuraya gidin: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Yeni bir OAuth 2.0 İstemci Kimliği oluşturun** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) --**"+ Kimlik Bilgilerini Oluştur"**→**"OAuth istemci kimliği"**seçeneğini tıklayın +**2. Create a new OAuth 2.0 Client ID** -- Uygulama türü:**"Web uygulaması"** -- Ad: beğendiğiniz herhangi bir şey (ör. 'OmniRoute Remote') +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -**3. Yetkili Yönlendirme URI'lerini Ekle** +**3. Add Authorized Redirect URIs** -**"Yetkili yönlendirme URI'leri"**alanına şunu ekleyin:``` +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> `Sunucunuz.com`u sunucunuzun alan adı veya IP'si ile değiştirin (gerekirse bağlantı noktasını ekleyin, örneğin `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. Kimlik bilgilerini kaydedin ve kopyalayın** +After creating, Google will show the **Client ID** and **Client Secret**. -Oluşturduktan sonra Google,**Müşteri Kimliğini**ve**Müşteri Sırrını**gösterecektir. +**5. Set environment variables** -**5. Ortam değişkenlerini ayarlama** +In your `.env` (or Docker environment variables): -`.env`nizde (veya Docker ortam değişkenlerinde):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1789,78 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. OmniRoute'u yeniden başlatın**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Tekrar bağlanmayı deneyin** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Kontrol Paneli → Sağlayıcılar → Antigravity (veya Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google artık doğru şekilde "https://sunucunuz.com/callback" adresine yönlendirecektir.--- +--- #### Temporary workaround (without custom credentials) -Şu anda kendi kimlik bilgilerinizi ayarlamak istemiyorsanız**manuel URL akışını**kullanmaya devam edebilirsiniz: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute, Google yetkilendirme URL'sini açar -2. Yetkilendirmeden sonra Google, "localhost"a yönlendirmeye çalışır (bu işlem uzak sunucuda başarısız olur) -3.**Tarayıcınızın adres çubuğundan tam URL'yi kopyalayın**(sayfa yüklenmese bile) -4. Bu URL'yi OmniRoute bağlantı modunda gösterilen alana yapıştırın -5.**"Bağlan"**'a tıklayın +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Bu, URL'deki yetkilendirme kodunun, yönlendirme sayfasının yüklenip yüklenmediğine bakılmaksızın geçerli olması nedeniyle işe yarar.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. - -🇧🇷 Portekiz Versiyonu#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Kimlik doğrulama için**Antigravity**ve**Gemini CLI**'nin**Google OAuth 2.0**kullandığı kanıtlanmıştır. Google, Google Cloud Console uygulaması için önceden hazırlanmış URI'ler nedeniyle**ekstra**OAuth akışı içermeyen bir "redirect_uri" kullanıyor. +
+🇧🇷 Versão em Português -OAuth kimlik doğrulaması, OmniRoute'un**localhost'ta**açılmasına izin vermez. Uzak bir sunucuda OmniRoute'a eriştiğinizde (ör. "https://omniroute.meuservidor.com") veya Google, com kimlik doğrulamasını reddetti:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Sunucunuzun URI'sı ile**OAuth 2.0 İstemci Kimliği**olmadan Google Cloud Console'u kesin olarak girin.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Google Cloud Console'a erişim** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -**2. Yeni bir OAuth 2.0 İstemci Kimliğiyle Ağlayın** +**2. Crie um novo OAuth 2.0 Client ID** --**"+ Kimlik Bilgileri Oluştur"**→**"OAuth istemci kimliği"**seçeneğini tıklayın +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -- Uygulama türü:**"Web uygulaması"** -- Ad: escolha qualquer adı (ör. `OmniRoute Remote`) +**3. Adicione as Authorized Redirect URIs** -**3. Yetkili Yönlendirme URI'leri Olarak Adicione** +No campo **"Authorized redirect URIs"**, adicione: -**"Yetkili yönlendirme URI'leri"**ek açıklaması yok:``` +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Sunucunuzun IP'si veya IP'si için "seu-servidor.com" yerine (gerekli port dahil, örneğin: "http://45.33.32.156:20128/callback"). +**4. Salve e copie as credenciais** -**4. Kimlik bilgileri olarak kaydedin ve kopyalayın** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Yazdıktan sonra Google,**Müşteri Kimliği**ve**Müşteri Sırrı**'nı arar. +**5. Configure as variáveis de ambiente** -**5. Ortam değişkenleri olarak yapılandırın** +No seu `.env` (ou nas variáveis de ambiente do Docker): -".env" dosyası yok (veya Docker ortamının değişkenleri):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1869,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. OmniRoute'u Yeniden Başlat**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute +``` -```` +**7. Tente conectar novamente** -**7. Yeni bağlantıların kurulması** +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Kontrol Paneli → Sağlayıcılar → Antigravity (veya Gemini CLI) → OAuth +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. -Agora veya Google, "https://seu-servidor.com/callback" adresine doğru şekilde yeniden yönlendiriliyor ve kimlik doğrulama işlevi yapılıyor.--- +--- #### Workaround temporário (sem configurar credenciais próprias) -Öncelikli olarak kimlik bilgilerine gerek duymazsanız,**URL kılavuzunu**kullanabilirsiniz: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. OmniRoute, Google'a yetkilendirilmiş bir URL kaydeder -2. Yetki verdiğinizde veya Google, "localhost"a yeniden yönlendirme yaptığında (uzak sunucu olmadığından) -3.**Tarayıcınızın son çubuğuna eksiksiz bir URL kopyalayın**(yeniden giriş sayfası olmadan) -4. OmniRoute'ta bağlantı yöntemi olarak görünmeyen bir URL var -5.**"Bağlan"ı**tıklayın +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) +4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute +5. Clique em **"Connect"** -> Bu geçici çözüm, URL'nin yetkilendirme kodu nedeniyle, yönlendirmeden veya yönlendirmeden bağımsız olarak geçerlidir.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1907,64 +2171,73 @@ Agora veya Google, "https://seu-servidor.com/callback" adresine doğru şekilde ## 🛠️ Tech Stack - -Teknoloji yığını ayrıntılarını genişletmek için tıklayın +
+Click to expand tech stack details --**Çalışma zamanı**: Node.js 18–22 LTS (⚠️ Node.js 24+**desteklenmez**— `better-sqlite3` yerel ikili dosyaları uyumlu değildir) --**Dil**: TypeScript 5.9 —**%100 TypeScript**`src/` ve `open-sse/` genelinde (v2.0'dan bu yana çekirdek modüllerde sıfır `herhangi`) --**Çerçeve**: Next.js 16 + React 19 + Tailwind CSS 4 --**Veritabanı**: LowDB (JSON) + SQLite (etki alanı durumu + proxy günlükleri + MCP denetimi + yönlendirme kararları) --**Şemalar**: Zod (MCP aracı G/Ç doğrulaması, API sözleşmeleri) --**Protokoller**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Akış**: Sunucu Tarafından Gönderilen Etkinlikler (SSE) --**Kimlik Doğrulama**: OAuth 2.0 (PKCE) + JWT + API Anahtarları + MCP Kapsamlı Yetkilendirme --**Test**: Node.js test çalıştırıcısı + Vitest (birim, entegrasyon, E2E dahil 900'den fazla test) --**CI/CD**: GitHub Eylemleri (otomatik npm yayınlama + yayınlandığında Docker Hub) --**Web sitesi**: [omniroute.online](https://omniroute.online) --**Paket**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Esneklik**: Devre kesici, üstel geri çekilme, gök gürültüsüne karşı sürü, TLS yanıltma, otomatik kombo kendi kendini iyileştirme
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Belgeler -| Belge | Açıklama | +| Document | Description | | ---------------------------------------------- | --------------------------------------------------- | -| [Kullanıcı Kılavuzu](docs/USER_GUIDE.md) | Sağlayıcılar, kombinasyonlar, CLI entegrasyonu, dağıtım | -| [API Referansı](docs/API_REFERENCE.md) | Örneklerle birlikte tüm uç noktalar | -| [MCP Sunucusu](open-sse/mcp-server/README.md) | 16 MCP aracı, IDE yapılandırmaları, Python/TS/Go istemcileri | -| [A2A Sunucusu](src/lib/a2a/README.md) | JSON-RPC 2.0 protokolü, beceriler, akış, görev yönetimi | -| [Otomatik Kombo Motoru](docs/auto-combo.md) | 6 faktörlü puanlama, mod paketleri, kendi kendini iyileştirme | -| [Sorun Giderme](docs/TROUBLESHOOTING.md) | Common problems and solutions | -| [Mimari](docs/ARCHITECTURE.md) | Sistem mimarisi ve iç bileşenleri | -| [Katkıda bulunuyor](CONTRIBUTING.md) | Geliştirme kurulumu ve yönergeleri | -| [OpenAPI Spesifikasyonu](docs/openapi.yaml) | OpenAPI 3.0 spesifikasyonu | -| [Güvenlik Politikası](SECURITY.md) | Güvenlik açığı raporlaması ve güvenlik uygulamaları | -| [VM Dağıtımı](docs/VM_DEPLOYMENT_GUIDE.md) | Kılavuzun tamamı: VM + nginx + Cloudflare kurulumu | -| [Özellikler Galerisi](docs/FEATURES.md) | Ekran görüntüleriyle görsel kontrol paneli turu | -| [Sürüm Kontrol Listesi](docs/RELEASE_CHECKLIST.md) | Yayın öncesi doğrulama adımları |--- +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute, birden fazla geliştirme aşamasında**210'dan fazla planlanmış özelliğe**sahiptir. İşte kilit alanlar: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Kategori | Planlanan Özellikler | Öne Çıkanlar | -| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------------------- | -| 🧠**Yönlendirme ve İstihbarat**| 25+ | En düşük gecikmeli yönlendirme, etiket tabanlı yönlendirme, kota ön kontrolü, P2C hesap seçimi | -| 🔒**Güvenlik ve Uyumluluk**| 20+ | SSRF sağlamlaştırma, kimlik bilgileri gizleme, uç nokta başına hız sınırı, yönetim anahtarı kapsamı | -| 📊**Gözlemlenebilirlik**| 15+ | OpenTelemetry entegrasyonu, gerçek zamanlı kota izleme, model başına maliyet takibi | -| 🔄**Sağlayıcı Entegrasyonları**| 20+ | Dinamik model kaydı, sağlayıcı bekleme süreleri, çoklu hesap Codex, Copilot kota ayrıştırma | -| ⚡**Performans**| 15+ | Çift önbellek katmanı, bilgi istemi önbelleği, yanıt önbelleği, akışı canlı tutma, toplu API | -| 🌐**Ekosistem**| 10+ | WebSocket API, yapılandırma çalışırken yeniden yükleme, dağıtılmış yapılandırma deposu, ticari mod |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**OpenCode Entegrasyonu**— OpenCode AI kodlama IDE'si için yerel sağlayıcı desteği -- 🔗**TRAE Entegrasyonu**— TRAE AI geliştirme çerçevesi için tam destek -- 📦**Toplu API**— Toplu istekler için eşzamansız toplu işleme -- 🎯**Etiket Tabanlı Yönlendirme**— İstekleri özel etiketlere ve meta verilere dayalı olarak yönlendirin -- 💰**En Düşük Maliyet Stratejisi**— Mevcut en ucuz sağlayıcıyı otomatik olarak seç +### 🔜 Coming Soon -> 📝 Tüm özellik spesifikasyonları [`belgeler/yeni-özellikler/`](belgeler/yeni-özellikler/)'de mevcuttur (217 ayrıntılı özellik)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1972,18 +2245,20 @@ OmniRoute, birden fazla geliştirme aşamasında**210'dan fazla planlanmış öz ### How to Contribute -1. Depoyu çatallayın -2. Özellik dalınızı oluşturun (`git checkout -b feature/amazing-feature`) -3. Değişikliklerinizi kaydedin (`git commit -m 'Muhteşem özellik ekle'`) -4. Şubeye itin ('git push Origin özelliği/inanılmaz özellik') -5. Çekme İsteği Açın +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Ayrıntılı yönergeler için [CONTRIBUTING.md](CONTRIBUTING.md) adresine bakın.### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1995,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Bu çatala ilham veren orijinal proje olan**[decolua](https://github.com/decolua)**tarafından**[9router](https://github.com/decolua/9router)**'a özel teşekkürler. OmniRoute, ek özellikler, çok modlu API'ler ve tam TypeScript yeniden yazımı ile bu inanılmaz temel üzerine inşa edilmiştir. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Bu JavaScript bağlantı noktasına ilham veren orijinal Go uygulaması**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**'ye özellikle teşekkür ederiz.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Lisans -MIT Lisansı - ayrıntılar için bkz. [LİSANS](LİSANS).--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/tr/docs/ARCHITECTURE.md b/docs/i18n/tr/docs/ARCHITECTURE.md index 7aba25395f..8d7e85f741 100644 --- a/docs/i18n/tr/docs/ARCHITECTURE.md +++ b/docs/i18n/tr/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Son güncelleme: 2026-03-28_## Executive Summary -OmniRoute, Next.js üzerine oluşturulmuş yerel bir AI yönlendirme ağ geçidi ve kontrol panelidir. -OpenAI uyumlu tek bir uç nokta (`/v1/*`) sağlar ve çeviri, geri dönüş, belirteç yenileme ve kullanım izleme özellikleriyle trafiği birden fazla yukarı akış sağlayıcısına yönlendirir. -Çekirdek yetenekler: +_Last updated: 2026-03-28_ -- CLI/araçlar için OpenAI uyumlu API yüzeyi (28 sağlayıcı) -- Sağlayıcı formatları arasında istek/yanıt çevirisi -- Model birleşik geri dönüşü (çoklu model sırası) -- Hesap düzeyinde geri dönüş (sağlayıcı başına çoklu hesap) -- OAuth + API anahtarı sağlayıcı bağlantı yönetimi -- '/v1/embeddings' yoluyla yerleştirme oluşturma (6 sağlayıcı, 9 model) -- '/v1/images/ Generations' aracılığıyla görüntü oluşturma (4 sağlayıcı, 9 model) -- Akıl yürütme modelleri için etiket ayrıştırmayı (`...`) düşünün -- OpenAI SDK uyumluluğu için yanıt temizliği -- Sağlayıcılar arası uyumluluk için rol normalleştirme (geliştirici→sistem, sistem→kullanıcı) -- Yapılandırılmış çıktı dönüşümü (json_schema → Gemini ResponseSchema) -- Sağlayıcılar, anahtarlar, takma adlar, kombinasyonlar, ayarlar ve fiyatlandırma için yerel kalıcılık -- Kullanım/maliyet takibi ve talep kaydı -- Çoklu cihaz/durum senkronizasyonu için isteğe bağlı bulut senkronizasyonu -- API erişim kontrolü için IP izin verilenler listesi/engellenenler listesi -- Bütçe yönetimini düşünmek (geçişli/otomatik/özel/uyarlanabilir) -- Küresel sistem istemi enjeksiyonu -- Oturum takibi ve parmak izi alma -- Sağlayıcıya özel profillerle hesap başına geliştirilmiş oran sınırlaması -- Sağlayıcı esnekliği için devre kesici modeli -- Mutex kilitlemeli, yıldırım önleyici sürü koruması -- İmza tabanlı istek tekilleştirme önbelleği -- Etki alanı katmanı: model kullanılabilirliği, maliyet kuralları, geri dönüş politikası, kilitleme politikası -- Etki alanı durumunun kalıcılığı (geri dönüşler, bütçeler, kilitlemeler, devre kesiciler için SQLite yazma önbelleği) -- Merkezi talep değerlendirmesi için politika motoru (kilitleme → bütçe → geri dönüş) -- p50/p95/p99 gecikme toplama ile telemetri isteği -- Uçtan uca izleme için Korelasyon Kimliği (X-Request-Id) -- API anahtarı başına devre dışı bırakma özelliğiyle uyumluluk denetiminin günlüğe kaydedilmesi -- LLM kalite güvencesi için değerlendirme çerçevesi -- Gerçek zamanlı devre kesici durumuna sahip Resilience UI kontrol paneli -- Modüler OAuth sağlayıcıları ('src/lib/oauth/providers/' altında 12 ayrı modül) +## Executive Summary -Birincil çalışma zamanı modeli: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- "src/app/api/\*" altındaki Next.js uygulama rotaları, hem kontrol paneli API'lerini hem de uyumluluk API'lerini uygular -- "src/sse/_" + "open-sse/_" içindeki paylaşılan bir SSE/yönlendirme çekirdeği, sağlayıcı yürütmeyi, çeviriyi, akışı, geri dönüşü ve kullanımı yönetir## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Yerel ağ geçidi çalışma zamanı -- Kontrol paneli yönetimi API'leri -- Sağlayıcı kimlik doğrulaması ve belirteç yenileme -- Çeviri ve SSE akışı isteyin -- Yerel durum + kullanım kalıcılığı -- İsteğe bağlı bulut senkronizasyonu düzenlemesi### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- `NEXT_PUBLIC_CLOUD_URL` arkasında bulut hizmeti uygulaması -- Sağlayıcı SLA'sı/yerel sürecin dışındaki kontrol düzlemi -- Harici CLI ikili dosyalarının kendileri (Claude CLI, Codex CLI, vb.)## Dashboard Surface (Current) +### Out of Scope -'Src/app/(dashboard)/dashboard/' altındaki ana sayfalar: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — hızlı başlangıç + sağlayıcıya genel bakış -- `/dashboard/endpoint` — uç nokta proxy'si + MCP + A2A + API uç noktası sekmeleri -- `/dashboard/providers` — sağlayıcı bağlantıları ve kimlik bilgileri -- `/dashboard/combos` — birleşik stratejiler, şablonlar, model yönlendirme kuralları -- `/dashboard/costs` — maliyet toplama ve fiyatlandırma görünürlüğü -- `/dashboard/analytics` — kullanım analizleri ve değerlendirmeler -- `/dashboard/limits` — kota/oran kontrolleri -- `/dashboard/cli-tools` — CLI'ye katılım, çalışma zamanı algılama, yapılandırma oluşturma -- `/dashboard/agents` — algılanan ACP aracıları + özel aracı kaydı -- `/dashboard/media` — resim/video/müzik oyun alanı -- `/dashboard/search-tools` — arama sağlayıcı testi ve geçmişi -- `/dashboard/health` — çalışma süresi, devre kesiciler, oran sınırları -- `/dashboard/logs` — istek/proxy/denetim/konsol günlükleri -- `/dashboard/settings` — sistem ayarları sekmeleri (genel, yönlendirme, birleşik varsayılanlar vb.) -- `/dashboard/api-manager` — API anahtarı yaşam döngüsü ve model izinleri## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Ana dizinler: +Main directories: -- Uyumluluk API'leri için `src/app/api/v1/*` ve `src/app/api/v1beta/*` -- yönetim/yapılandırma API'leri için `src/app/api/*` -- Sonraki `next.config.mjs` haritasında `/v1/*` ile `/api/v1/*` arasında yeniden yazar +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Önemli uyumluluk yolları: +Important compatibility routes: -- 'src/app/api/v1/chat/completions/route.ts' +- `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` — 'custom: true' özelliğine sahip özel modelleri içerir -- `src/app/api/v1/embeddings/route.ts` — yerleştirme nesli (6 sağlayıcı) -- `src/app/api/v1/images/ Generations/route.ts` — görüntü oluşturma (Antigravity/Nebius dahil 4+ sağlayıcı) -- `src/app/api/v1/messages/count_tokens/route.ts' -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — sağlayıcı başına özel sohbet -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — sağlayıcı başına özel yerleştirmeler -- `src/app/api/v1/providers/[provider]/images/ Generations/route.ts` — sağlayıcı başına ayrılmış görüntüler -- 'src/app/api/v1beta/models/route.ts' -- `src/app/api/v1beta/models/[...yol]/route.ts' +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) +- `src/app/api/v1/messages/count_tokens/route.ts` +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images +- `src/app/api/v1beta/models/route.ts` +- `src/app/api/v1beta/models/[...path]/route.ts` -Yönetim alanları: +Management domains: -- Kimlik doğrulama/ayarlar: "src/app/api/auth/_", "src/app/api/settings/_" -- Sağlayıcılar/bağlantılar: `src/app/api/providers\*' -- Sağlayıcı düğümleri: `src/app/api/provider-nodes\*' -- Özel modeller: `src/app/api/provider-models` (GET/POST/DELETE) -- Model kataloğu: `src/app/api/models/route.ts` (GET) -- Proxy yapılandırması: "src/app/api/settings/proxy" (GET/PUT/DELETE) + "src/app/api/settings/proxy/test" (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- Anahtarlar/takma adlar/kombinasyonlar/fiyatlandırma: "src/app/api/keys*", "src/app/api/models/alias", "src/app/api/combos*", "src/app/api/pricing" -- Kullanım: `src/app/api/usage/*` -- Senkronizasyon/bulut: "src/app/api/sync/_", "src/app/api/cloud/_" -- CLI araç yardımcıları: `src/app/api/cli-tools/*` -- IP filtresi: `src/app/api/settings/ip-filter` (GET/PUT) -- Düşünme bütçesi: `src/app/api/settings/thinking-budget` (GET/PUT) -- Sistem istemi: `src/app/api/settings/system-prompt` (GET/PUT) -- Oturumlar: `src/app/api/sessions` (GET) -- Hız sınırları: `src/app/api/rate-limits` (GET) -- Esneklik: "src/app/api/resilience" (GET/PATCH) — sağlayıcı profilleri, devre kesici, hız sınırı durumu -- Dayanıklılığı sıfırlama: `src/app/api/resilience/reset` (POST) — kesicileri + bekleme sürelerini sıfırla -- Önbellek istatistikleri: `src/app/api/cache/stats` (GET/DELETE) -- Model kullanılabilirliği: "src/app/api/models/availability" (GET/POST) -- Telemetri: 'src/app/api/telemetry/summary' (GET) -- Bütçe: `src/app/api/usage/budget` (GET/POST) -- Geri dönüş zincirleri: `src/app/api/fallback/chains' (GET/POST/DELETE) -- Uyumluluk denetimi: `src/app/api/compliance/audit-log` (GET) -- Değerlendirmeler: "src/app/api/evals" (GET/POST), "src/app/api/evals/[suiteId]" (GET) -- Politikalar: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -Ana akış modülleri: +## 2) SSE + Translation Core -- Giriş: `src/sse/handlers/chat.ts` -- Çekirdek düzenleme: `open-sse/handlers/chatCore.ts` -- Sağlayıcı yürütme bağdaştırıcıları: `open-sse/executors/\*' -- Biçim algılama/sağlayıcı yapılandırması: `open-sse/services/provider.ts` -- Model ayrıştırma/çözme: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Hesap geri dönüş mantığı: `open-sse/services/accountFallback.ts` -- Çeviri kaydı: `open-sse/translator/index.ts` -- Akış dönüşümleri: "open-sse/utils/stream.ts", "open-sse/utils/streamHandler.ts" -- Kullanım çıkarma/normalleştirme: `open-sse/utils/usageTracking.ts` -- Etiket ayrıştırıcıyı düşünün: `open-sse/utils/thinkTagParser.ts` -- Gömme işleyicisi: 'open-sse/handlers/embeddings.ts' -- Katıştırma sağlayıcısı kayıt defteri: `open-sse/config/embeddingRegistry.ts` -- Görüntü oluşturma işleyicisi: `open-sse/handlers/imageGeneration.ts` -- Resim sağlayıcı kaydı: `open-sse/config/imageRegistry.ts` -- Yanıt temizleme: `open-sse/handlers/responseSanitizer.ts` -- Rol normalleştirme: `open-sse/services/roleNormalizer.ts` +Main flow modules: -Hizmetler (iş mantığı): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- Hesap seçimi/puanlama: `open-sse/services/accountSelector.ts` -- Bağlam yaşam döngüsü yönetimi: 'open-sse/services/contextManager.ts' -- IP filtresi uygulaması: `open-sse/services/ipFilter.ts` -- Oturum izleme: `open-sse/services/sessionManager.ts` -- Tekilleştirme isteği: `open-sse/services/signatureCache.ts` -- Sistem istemi enjeksiyonu: 'open-sse/services/systemPrompt.ts' -- Bütçe yönetimini düşünmek: `open-sse/services/thinkingBudget.ts` -- Joker karakterli model yönlendirme: `open-sse/services/wildcardRouter.ts` -- Oran sınırı yönetimi: `open-sse/services/rateLimitManager.ts` -- Devre kesici: 'open-sse/services/circuitBreaker.ts' +Services (business logic): -Etki alanı katmanı modülleri: +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- Model kullanılabilirliği: `src/lib/domain/modelAvailability.ts` -- Maliyet kuralları/bütçeler: `src/lib/domain/costRules.ts` -- Geri dönüş politikası: `src/lib/domain/fallbackPolicy.ts` -- Birleşik çözümleyici: `src/lib/domain/comboResolver.ts` -- Kilitleme politikası: `src/lib/domain/lockoutPolicy.ts` -- Politika motoru: `src/domain/policyEngine.ts` — merkezi kilitleme → bütçe → geri dönüş değerlendirmesi -- Hata kodları kataloğu: `src/lib/domain/errorCodes.ts` -- İstek Kimliği: `src/lib/domain/requestId.ts` -- Getirme zaman aşımı: `src/lib/domain/fetchTimeout.ts` -- Telemetri isteği: `src/lib/domain/requestTelemetry.ts` -- Uyumluluk/denetim: `src/lib/domain/compliance/index.ts` -- Değerlendirme çalıştırıcısı: `src/lib/domain/evalRunner.ts` -- Etki alanı durumu kalıcılığı: `src/lib/db/domainState.ts` — Geri dönüş zincirleri, bütçeler, maliyet geçmişi, kilitleme durumu, devre kesiciler için SQLite CRUD +Domain layer modules: -OAuth sağlayıcı modülleri ('src/lib/oauth/providers/' altında 12 ayrı dosya): +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -- Kayıt defteri dizini: `src/lib/oauth/providers/index.ts` -- Bireysel sağlayıcılar: "claude.ts", "codex.ts", "gemini.ts", "antigravity.ts", "qoder.ts", "qwen.ts", "kimi-coding.ts", "github.ts", "kiro.ts", "cursor.ts", "kilocode.ts", 'cline.ts' -- İnce sarmalayıcı: `src/lib/oauth/providers.ts` — bireysel modüllerden yeniden dışa aktarım## 3) Persistence Layer +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -Birincil durum DB'si (SQLite): +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -- Çekirdek altyapısı: `src/lib/db/core.ts` (better-sqlite3, geçişler, WAL) -- Dışa aktarma cephesi: `src/lib/localDb.ts` (arayanlar için ince uyumluluk katmanı) -- dosya: `${DATA_DIR}/storage.sqlite` (veya ayarlandığında `$XDG_CONFIG_HOME/omniroute/storage.sqlite`, aksi takdirde `~/.omniroute/storage.sqlite`) -- varlıklar (tablolar + KV ad alanları): sağlayıcı Bağlantıları, sağlayıcı Node'ları, modelAliases, kombinasyonlar, apiKey'ler, ayarlar, fiyatlandırma,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +## 3) Persistence Layer -Kullanım kalıcılığı: +Primary state DB (SQLite): -- cephe: `src/lib/usageDb.ts` (`src/lib/usage/*` içinde ayrıştırılmış modüller) -- `storage.sqlite` içindeki SQLite tabloları: `usage_history`, `call_logs`, `proxy_logs` -- uyumluluk/hata ayıklama için isteğe bağlı dosya yapıları kalır (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- eski JSON dosyaları, mevcut olduklarında başlangıç geçişleriyle SQLite'a taşınır +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -Etki Alanı Durumu Veritabanı (SQLite): +Usage persistence: -- `src/lib/db/domainState.ts` — Etki alanı durumu için CRUD işlemleri -- Tablolar ("src/lib/db/core.ts" içinde oluşturulmuştur): "domain_fallback_chains", "domain_budgets", "domain_cost_history", "domain_lockout_state", "domain_circuit_breakers" -- İçe yazma önbellek modeli: bellek içi Haritalar çalışma zamanında yetkilidir; mutasyonlar SQLite'a eşzamanlı olarak yazılır; soğuk başlatma sırasında durum DB'den geri yüklenir## 4) Auth + Security Surfaces +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- Kontrol paneli çerez kimlik doğrulaması: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- API anahtarı oluşturma/doğrulama: `src/shared/utils/apiKey.ts` -- Sağlayıcı sırları 'providerConnections' girişlerinde kalıcı oldu -- 'open-sse/utils/proxyFetch.ts' (env vars) ve 'open-sse/utils/networkProxy.ts' (sağlayıcı başına yapılandırılabilir veya genel) aracılığıyla giden proxy desteği## 5) Cloud Sync +Domain State DB (SQLite): -- Zamanlayıcı başlatma: "src/lib/initCloudSync.ts", "src/shared/services/initializeCloudSync.ts", "src/shared/services/modelSyncScheduler.ts" -- Periyodik görev: `src/shared/services/cloudSyncScheduler.ts` -- Periyodik görev: `src/shared/services/modelSyncScheduler.ts` -- Kontrol rotası: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Geri dönüş kararları, durum kodları ve hata mesajı buluşsal yöntemleri kullanılarak "open-sse/services/accountFallback.ts" tarafından yönlendirilir. Birleşik yönlendirme ekstra bir koruma ekler: yukarı akış içerik bloğu ve rol doğrulama hataları gibi sağlayıcı kapsamlı 400'ler, model yerel hataları olarak ele alınır, böylece daha sonraki birleşik hedefler çalışmaya devam edebilir.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -Canlı trafik sırasında yenileme, 'refreshCredentials()' yürütücüsü aracılığıyla 'open-sse/handlers/chatCore.ts' içinde yürütülür.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -Bulut etkinleştirildiğinde periyodik senkronizasyon "CloudSyncScheduler" tarafından tetiklenir.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -Fiziksel depolama dosyaları: +Physical storage files: -- birincil çalışma zamanı veritabanı: `${DATA_DIR}/storage.sqlite` -- istek günlük satırları: `${DATA_DIR}/log.txt` (compat/debug yapısı) -- yapılandırılmış çağrı verisi arşivleri: `${DATA_DIR}/call_logs/` -- isteğe bağlı çevirmen/hata ayıklama isteğinde bulunma oturumları: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: uyumluluk API'leri -- `src/app/api/v1/providers/[provider]/*`: sağlayıcı başına ayrılmış yollar (sohbet, yerleştirmeler, resimler) -- `src/app/api/providers*`: sağlayıcı CRUD, doğrulama, test etme -- `src/app/api/provider-nodes*`: özel uyumlu düğüm yönetimi -- `src/app/api/provider-models`: özel model yönetimi (CRUD) -- `src/app/api/models/route.ts`: model kataloğu API'si (takma adlar + özel modeller) -- `src/app/api/oauth/*`: OAuth/cihaz kodu akışları -- `src/app/api/keys*`: yerel API anahtarı yaşam döngüsü -- `src/app/api/models/alias`: takma ad yönetimi -- `src/app/api/combos*`: geri dönüş kombo yönetimi -- `src/app/api/pricing`: maliyet hesaplaması için fiyatlandırmayı geçersiz kılma -- `src/app/api/settings/proxy`: proxy yapılandırması (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: giden proxy bağlantı testi (POST) -- `src/app/api/usage/*`: kullanım ve günlük API'leri -- `src/app/api/sync/*` + `src/app/api/cloud/*`: bulut senkronizasyonu ve buluta yönelik yardımcılar -- `src/app/api/cli-tools/*`: yerel CLI yapılandırma yazarları/denetleyicileri -- `src/app/api/settings/ip-filter`: IP izin verilenler listesi/engellenenler listesi (GET/PUT) -- `src/app/api/settings/thinking-budget`: düşünme belirteci bütçe yapılandırması (GET/PUT) -- `src/app/api/settings/system-prompt`: genel sistem istemi (GET/PUT) -- `src/app/api/sessions`: aktif oturum listesi (GET) -- `src/app/api/rate-limits`: hesap başına oran sınırı durumu (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: istek ayrıştırma, birleşik işleme, hesap seçim döngüsü -- `open-sse/handlers/chatCore.ts`: çeviri, yürütücü gönderimi, yeniden deneme/yenileme işlemi, akış kurulumu -- `open-sse/executors/*`: sağlayıcıya özel ağ ve format davranışı### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: çevirmen kaydı ve orkestrasyonu -- Çevirmen iste: `open-sse/translator/request/\*' -- Yanıt çevirmenleri: `open-sse/translator/response/\*' -- Biçim sabitleri: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: SQLite'ta kalıcı yapılandırma/durum ve etki alanı kalıcılığı -- `src/lib/localDb.ts`: Veritabanı modülleri için uyumluluğun yeniden dışa aktarımı -- `src/lib/usageDb.ts`: SQLite tablolarının üstünde kullanım geçmişi/çağrı günlükleri görünümü## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Her sağlayıcının, URL oluşturma, başlık oluşturma, üstel geri alma ile yeniden deneme, kimlik bilgisi yenileme kancaları ve "execute()" düzenleme yöntemini sağlayan "BaseExecutor"u ("open-sse/executors/base.ts" içinde) genişleten özel bir yürütücüsü vardır. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Yürütücü | Sağlayıcı(lar) | Özel İşleme | -| -------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------- | -| 'DefaultExecutor' | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Şaşkınlık, Birlikte, Havai Fişek, Cerebras, Cohere, NVIDIA | Sağlayıcı başına dinamik URL/başlık yapılandırması | -| 'Antiyerçekimi Yürütücüsü' | Google Yerçekimine Karşı | Özel proje/oturum kimlikleri, Ayrıştırmadan Sonra Yeniden Dene | -| 'CodexExecutor' | OpenAI Kodeksi | Sistem talimatlarını enjekte ediyor, muhakeme çabasını zorluyor | -| 'İmleçYürütücüsü' | İmleç IDE'si | ConnectRPC protokolü, Protobuf kodlaması, sağlama toplamı yoluyla imzalama isteği | -| 'GithubYürütücüsü' | GitHub Yardımcı Pilotu | Yardımcı Pilot belirteci yenilemesi, VSCode'u taklit eden başlıklar | -| 'KiroExecutor' | AWS CodeWhisperer/Kiro | AWS EventStream ikili biçimi → SSE dönüşümü | -| `GeminiCLIExecutor` | İkizler CLI | Google OAuth jetonu yenileme döngüsü | +### Persistence -Diğer tüm sağlayıcılar (özel uyumlu düğümler dahil) 'DefaultExecutor'u kullanır.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Sağlayıcı | Biçim | Yetki | Akış | Yayın Dışı | Jeton Yenileme | Kullanım API'si | -| ---------------------- | ----------------- | ----------------------------- | ------------------- | ---------- | -------------- | ------------------------- | ------------------------------ | -| Claude | Claude | API Anahtarı / OAuth | ✅ | ✅ | ✅ | ⚠️ Yalnızca yönetici | -| İkizler | ikizler | API Anahtarı / OAuth | ✅ | ✅ | ✅ | ⚠️ Bulut Konsolu | -| İkizler CLI | İkizler-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Bulut Konsolu | -| Yer çekimine karşı | yerçekimine karşı | OAuth | ✅ | ✅ | ✅ | ✅ Tam kota API'si | -| OpenAI | açık | API Anahtarı | ✅ | ✅ | ❌ | ❌ | -| Kodeks | openai-yanıtları | OAuth | ✅ zorunlu | ❌ | ✅ | ✅ Oran sınırları | -| GitHub Yardımcı Pilotu | açık | OAuth + Yardımcı Pilot Jetonu | ✅ | ✅ | ✅ | ✅ Kota anlık görüntüleri | -| İmleç | imleç | Özel sağlama toplamı | ✅ | ✅ | ❌ | ❌ | -| Kiro | kira | AWS SSO OIDC | ✅ (Etkinlik Akışı) | ❌ | ✅ | ✅ Kullanım sınırları | -| Qwen | açık | OAuth | ✅ | ✅ | ✅ | ⚠️ İsteğe göre | -| Kod | açık | OAuth (Temel) | ✅ | ✅ | ✅ | ⚠️ İsteğe göre | -| OpenRouter | açık | API Anahtarı | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | Claude | API Anahtarı | ✅ | ✅ | ❌ | ❌ | -| Derin Arama | açık | API Anahtarı | ✅ | ✅ | ❌ | ❌ | -| Büyük | açık | API Anahtarı | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | açık | API Anahtarı | ✅ | ✅ | ❌ | ❌ | -| Mistral | açık | API Anahtarı | ✅ | ✅ | ❌ | ❌ | -| Şaşkınlık | açık | API Anahtarı | ✅ | ✅ | ❌ | ❌ | -| Birlikte AI | açık | API Anahtarı | ✅ | ✅ | ❌ | ❌ | -| Havai Fişek Yapay Zeka | açık | API Anahtarı | ✅ | ✅ | ❌ | ❌ | -| Beyinler | açık | API Anahtarı | ✅ | ✅ | ❌ | ❌ | -| Tutarlı | açık | API Anahtarı | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | açık | API Anahtarı | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Algılanan kaynak formatları şunları içerir: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. -- "açık" -- "açık yanıtlar" -- 'Claude' -- 'ikizler' +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | -Hedef formatlar şunları içerir: +All other providers (including custom compatible nodes) use the `DefaultExecutor`. -- OpenAI sohbeti/Yanıtlar - -Claude -- Gemini/Gemini-CLI/Antiyer çekimi zarfı +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: + +- `openai` +- `openai-responses` +- `claude` +- `gemini` + +Target formats include: + +- OpenAI chat/Responses +- Claude +- Gemini/Gemini-CLI/Antigravity envelope - Kiro -- İmleç +- Cursor -Çeviriler, hub formatı olarak**OpenAI**kullanır; tüm dönüşümler, ara düzey olarak OpenAI üzerinden gerçekleştirilir:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Çeviriler, kaynak yükünün şekline ve sağlayıcının hedef biçimine göre dinamik olarak seçilir. +Additional processing layers in the translation pipeline: -Çeviri hattındaki ek işleme katmanları: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Yanıt temizleme**— Kesin SDK uyumluluğunu sağlamak için standart olmayan alanları OpenAI biçimindeki yanıtlardan (hem akışlı hem de akışsız) çıkarır --**Rol normalleştirme**— OpenAI olmayan hedefler için "geliştirici" → "sistem"i dönüştürür; sistem rolünü reddeden modeller için "sistem" → "kullanıcı"yı birleştirir (GLM, ERNIE) --**Think etiketi çıkarma**— İçerikteki "..." bloklarını "reasoning_content" alanına ayrıştırır --**Yapılandırılmış çıktı**— OpenAI `response_format.json_schema`yı Gemini'nin `responseMimeType` + `responseSchema`sına dönüştürür## Supported API Endpoints +## Supported API Endpoints -| Uç nokta | Biçim | İşleyici | +| Endpoint | Format | Handler | | -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | -| 'POST /v1/sohbet/tamamlamalar' | OpenAI Sohbet | `src/sse/handlers/chat.ts` | -| 'POST /v1/mesajlar' | Claude Mesajları | Aynı işleyici (otomatik olarak algılandı) | -| 'POST /v1/yanıtlar' | OpenAI Yanıtları | 'open-sse/handlers/responsesHandler.ts' | -| 'POST /v1/yerleştirmeler' | OpenAI Yerleştirmeleri | 'open-sse/handlers/embeddings.ts' | -| 'GET /v1/yerleştirmeler' | Model listesi | API rotası | -| 'POST /v1/images/nesiller' | OpenAI Resimleri | 'open-sse/handlers/imageGeneration.ts' | -| 'GET /v1/images/nesiller' | Model listesi | API rotası | -| `POST /v1/providers/{provider}/chat/tamamlamalar` | OpenAI Sohbet | Model doğrulamayla sağlayıcı başına özel | -| `POST /v1/providers/{provider}/embeddings` | OpenAI Yerleştirmeleri | Model doğrulamayla sağlayıcı başına özel | -| `POST /v1/providers/{provider}/images/jenerasyonlar` | OpenAI Resimleri | Model doğrulamayla sağlayıcı başına özel | -| 'POST /v1/messages/count_tokens' | Claude Token Sayımı | API rotası | -| 'GET /v1/models' | OpenAI Modelleri listesi | API rotası (sohbet + yerleştirme + resim + özel modeller) | -| 'GET /api/models/catalog' | Katalog | Tüm modeller sağlayıcı + türe göre gruplandırılmıştır | -| 'POST /v1beta/models/*:streamGenerateContent' | İkizler yerlisi | API rotası | -| 'GET/PUT/DELETE /api/settings/proxy' | Proxy Yapılandırması | Ağ proxy yapılandırması | -| 'POST /api/settings/proxy/test' | Proxy Bağlantısı | Proxy durumu/bağlantı testi uç noktası | -| 'GET/POST/DELETE /api/provider-models' | Sağlayıcı Modelleri | Özel ve yönetilen mevcut modelleri destekleyen sağlayıcı modeli meta verileri |## Bypass Handler +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -Atlama işleyicisi (`open-sse/utils/bypassHandler.ts`), Claude CLI'den gelen bilinen "tek kullanımlık" istekleri (ısınma pingleri, başlık çıkarmalar ve belirteç sayıları) yakalar ve yukarı akış sağlayıcı belirteçlerini tüketmeden bir**sahte yanıt**döndürür. Bu yalnızca "Kullanıcı Aracısı" "claude-cli" içerdiğinde tetiklenir.## Request Logger Pipeline +## Bypass Handler -İstek kaydedici ('open-sse/utils/requestLogger.ts'), varsayılan olarak devre dışı bırakılan ve 'ENABLE_REQUEST_LOGS=true' aracılığıyla etkinleştirilen 7 aşamalı bir hata ayıklama günlük kaydı ardışık düzeni sağlar:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Dosyalar her istek oturumu için `/logs//` dizinine yazılır.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- geçici/oran/kimlik hatalarında sağlayıcı hesabı bekleme süresi -- başarısız istekten önce hesap geri dönüşü -- geçerli model/sağlayıcı yolu tükendiğinde birleşik model geri dönüşü## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- yenilenebilir sağlayıcılar için ön kontrol ve yenileme ile yeniden deneme -- Çekirdek yolda yenileme denemesinden sonra 401/403 yeniden deneme## 3) Stream Safety +## 2) Token Expiry -- bağlantının kesilmesine duyarlı akış denetleyicisi -- yayın sonu temizleme ve "[BİTTİ]" işlemeli çeviri akışı -- sağlayıcı kullanım meta verileri eksik olduğunda kullanım tahmini geri dönüşü## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- senkronizasyon hataları ortaya çıkıyor ancak yerel çalışma zamanı devam ediyor -- zamanlayıcının yeniden deneme özellikli mantığı vardır, ancak periyodik yürütme şu anda varsayılan olarak tek denemeli senkronizasyonu çağırır## 5) Data Integrity +## 3) Stream Safety -- Başlangıçta SQLite şema geçişleri ve otomatik yükseltme kancaları -- eski JSON → SQLite geçiş uyumluluk yolu## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Çalışma zamanı görünürlük kaynakları: +## 4) Cloud Sync Degradation -- `src/sse/utils/logger.ts` adresinden konsol günlükleri -- SQLite'ta istek başına kullanım toplamları ("usage_history", "call_logs", "proxy_logs") -- `settings.detailed_logs_enabled=true` olduğunda SQLite'da ("request_detail_logs") dört aşamalı ayrıntılı yük yakalamaları -- 'log.txt' dosyasındaki metinsel istek durumu günlüğü (isteğe bağlı/uyumlu) -- "ENABLE_REQUEST_LOGS=true" olduğunda "logs/" altında isteğe bağlı derin istek/çeviri günlükleri -- kullanıcı arayüzü tüketimi için kontrol paneli kullanım uç noktaları (`/api/usage/*`) +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -Ayrıntılı istek yükü yakalama, yönlendirilen çağrı başına en fazla dört JSON verisi aşamasını saklar: +## 5) Data Integrity -- istemciden alınan ham istek -- çevrilmiş istek aslında yukarı yönde gönderildi -- sağlayıcı yanıtı JSON olarak yeniden yapılandırıldı; akışlı yanıtlar, son özet artı akış meta verilerine sıkıştırılır -- OmniRoute tarafından döndürülen son müşteri yanıtı; akışlı yanıtlar aynı kompakt özet formunda saklanır## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- JWT sırrı ("JWT_SECRET"), kontrol paneli oturumu çerez doğrulamasını/imzalamayı güvence altına alır -- İlk parola önyüklemesi ("INITIAL_PASSWORD"), ilk çalıştırma yetkilendirmesi için açıkça yapılandırılmalıdır -- API anahtarı HMAC sırrı (`API_KEY_SECRET`), oluşturulan yerel API anahtarı biçimini korur -- Sağlayıcı sırları (API anahtarları/belirteçleri) yerel veritabanında kalıcıdır ve dosya sistemi düzeyinde korunmalıdır -- Bulut senkronizasyonu uç noktaları, API anahtarı kimlik doğrulaması + makine kimliği semantiğine dayanır## Environment and Runtime Matrix +## Observability and Operational Signals -Kod tarafından aktif olarak kullanılan ortam değişkenleri: +Runtime visibility sources: -- Uygulama/kimlik doğrulama: `JWT_SECRET`, `INITIAL_PASSWORD` -- Depolama: `DATA_DIR` -- Uyumlu düğüm davranışı: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- İsteğe bağlı depolama tabanını geçersiz kılma ('DATA_DIR' ayarlanmadığında Linux/macOS): 'XDG_CONFIG_HOME' -- Güvenlik karması: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Günlüğe kaydetme: `ENABLE_REQUEST_LOGS` -- Senkronizasyon/bulut URL'si: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- Giden proxy: "HTTP_PROXY", "HTTPS_PROXY", "ALL_PROXY", "NO_PROXY" ve küçük harf çeşitleri -- SOCKS5 özellik işaretleri: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- Platform/çalışma zamanı yardımcıları (uygulamaya özel yapılandırma değil): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` ve `localDb` eski dosya geçişiyle aynı temel dizin politikasını (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) paylaşır. -2. `/api/v1/route.ts`, anlamsal kaymayı önlemek için `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) tarafından kullanılan aynı birleşik katalog oluşturucuya yetki verir. -3. İstek kaydedici etkinleştirildiğinde tüm başlıkları/gövdeyi yazar; günlük dizinini hassas olarak değerlendirin. -4. Bulut davranışı, doğru `NEXT_PUBLIC_BASE_URL`ye ve bulut uç noktası erişilebilirliğine bağlıdır. -5. `open-sse/` dizini `@omniroute/open-sse`**npm çalışma alanı paketi**olarak yayınlanır. Kaynak kodu bunu `@omniroute/open-sse/...` aracılığıyla içe aktarır (Next.js `transpilePackages` tarafından çözümlenir). Bu belgedeki dosya yolları tutarlılık açısından hâlâ `open-sse/` dizin adını kullanıyor. -6. Kontrol panelindeki grafikler, erişilebilir, etkileşimli analiz görselleştirmeleri (model kullanım çubuk grafikleri, başarı oranlarını içeren sağlayıcı döküm tabloları) için**Recharts**(SVG tabanlı) kullanır. -7. E2E testleri**Oyun Yazarı**("tests/e2e/`) kullanır ve 'npm run test:e2e' aracılığıyla yürütülür. Birim testleri**Node.js test çalıştırıcısını**("tests/unit/`) kullanır ve "npm run test:unit" aracılığıyla çalıştırılır. `src/` altındaki kaynak kodu**TypeScript**'tir (`.ts`/`.tsx`); `open-sse/` çalışma alanı JavaScript (`.js`) olarak kalır. -8. Ayarlar sayfası 5 sekme halinde düzenlenmiştir: Güvenlik, Yönlendirme (6 genel strateji: önce doldurma, hepsini bir kez deneme, p2c, rastgele, en az kullanılan, maliyet açısından optimize edilmiş), Dayanıklılık (düzenlenebilir hız sınırları, devre kesici, politikalar), AI (bütçeyi düşünme, sistem istemi, istem önbelleği), Gelişmiş (proxy).## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- Kaynaktan derle: `npm run build' -- Docker görüntüsü oluşturun: `docker build -t omniroute.' -- Hizmeti başlatın ve şunları doğrulayın: -- '/api/ayarları AL' -- '/api/v1/models' AL' -- `PORT=20128` olduğunda CLI hedefi temel URL'si `http://:20128/v1` olmalıdır +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` +- `GET /api/v1/models` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/tr/docs/FEATURES.md b/docs/i18n/tr/docs/FEATURES.md index aab07c96cc..0fbbe447ef 100644 --- a/docs/i18n/tr/docs/FEATURES.md +++ b/docs/i18n/tr/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -OmniRoute panosunun her bölümüne yönelik görsel kılavuz.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Yapay zeka sağlayıcı bağlantılarını yönetin: OAuth sağlayıcıları (Claude Code, Codex, Gemini CLI), API anahtarı sağlayıcıları (Groq, DeepSeek, OpenRouter) ve ücretsiz sağlayıcılar (Qoder, Qwen, Kiro). Kiro hesapları, kredi bakiyesi takibini içerir; kalan krediler, toplam ödenek ve Kontrol Paneli → Kullanım bölümünde görünen yenileme tarihi.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -6 stratejiyle model yönlendirme kombinasyonları oluşturun: öncelikli, ağırlıklı, hepsini bir kez deneme, rastgele, en az kullanılan ve maliyet açısından optimize edilmiş. Her bir kombinasyon birden fazla modeli otomatik geri dönüşle zincirler ve hızlı şablonlar ve hazırlık kontrolleri içerir.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Belirteç tüketimi, maliyet tahminleri, etkinlik ısı haritaları, haftalık dağıtım grafikleri ve sağlayıcı başına dökümler içeren kapsamlı kullanım analitiği.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Gerçek zamanlı izleme: çalışma süresi, bellek, sürüm, gecikme yüzdeleri (p50/p95/p99), önbellek istatistikleri ve sağlayıcı devre kesici durumları.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -API çevirilerinde hata ayıklamaya yönelik dört mod:**Oyun Alanı**(format dönüştürücü),**Sohbet Test Aracı**(canlı istekler),**Test Bench**(toplu testler) ve**Canlı Monitör**(gerçek zamanlı akış).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Herhangi bir modeli doğrudan kontrol panelinden test edin. Sağlayıcıyı, modeli ve uç noktayı seçin, Monaco Düzenleyici ile istemler yazın, yanıtları gerçek zamanlı olarak yayınlayın, akışın ortasında iptal edin ve zamanlama ölçümlerini görüntüleyin.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Kontrol panelinin tamamı için özelleştirilebilir renk temaları. 7 ön ayarlı renk arasından seçim yapın (Mercan, Mavi, Kırmızı, Yeşil, Menekşe, Turuncu, Camgöbeği) veya altıgen renklerden herhangi birini seçerek özel bir tema oluşturun. Açık, karanlık ve sistem modunu destekler.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Sekmeli kapsamlı ayarlar paneli: +Comprehensive settings panel with tabs: --**Genel**— Sistem depolama, yedekleme yönetimi (veritabanını dışa/içe aktarma) -**Görünüm**— Tema seçici (koyu/açık/sistem), renk teması ön ayarları ve özel renkler, sağlık günlüğü görünürlüğü, kenar çubuğu öğesi görünürlük kontrolleri -**Güvenlik**— API uç nokta koruması, özel sağlayıcı engelleme, IP filtreleme, oturum bilgileri -**Yönlendirme**— Model takma adları, arka plan görevinin bozulması -**Esneklik**— Hız sınırı kalıcılığı, devre kesici ayarı, yasaklı hesapları otomatik olarak devre dışı bırakma, sağlayıcının son kullanma tarihi izleme -**Gelişmiş**— Yapılandırma geçersiz kılmaları, yapılandırma denetim takibi, geri dönüş bozulma modu![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Yapay zeka kodlama araçları için tek tıklamayla yapılandırma: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor ve Factory Droid. Otomatik yapılandırma uygulama/sıfırlama, bağlantı profilleri ve model eşleme özelliklerine sahiptir.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -CLI aracılarını keşfetmeye ve yönetmeye yönelik kontrol paneli. Aşağıdakilerle birlikte 14 yerleşik aracıdan (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) oluşan bir tabloyu gösterir: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Yükleme durumu**— Sürüm algılamayla Kurulu / Bulunamadı -**Protokol rozetleri**— stdio, HTTP vb. -**Özel aracılar**— Herhangi bir CLI aracını form aracılığıyla kaydedin (ad, ikili dosya, sürüm komutu, ortaya çıkan argümanlar) -**CLI Parmak İzi Eşleştirme**— Sağlayıcı başına yerel CLI istek imzalarını eşleştirmek için geçiş yaparak proxy IP'yi korurken yasak riskini azaltır--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Kontrol panelinden resimler, videolar ve müzik oluşturun. OpenAI, xAI, Together, Hyperbolik, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open ve MusicGen'i destekler.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Sağlayıcıya, modele, hesaba ve API anahtarına göre filtrelemeyle gerçek zamanlı istek günlüğü kaydı. Durum kodlarını, belirteç kullanımını, gecikmeyi ve yanıt ayrıntılarını gösterir.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Yetenek dökümüyle birleştirilmiş API uç noktanız: Sohbet Tamamlamalar, Yanıtlar API'si, Yerleştirmeler, Görüntü Oluşturma, Yeniden Sıralama, Ses Transkripsiyon, Metinden Konuşmaya, Denetlemeler ve kayıtlı API anahtarları. Uzaktan erişim için Cloudflare Hızlı Tünel entegrasyonu ve bulut proxy desteği.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -API anahtarlarını oluşturun, kapsamını belirleyin ve iptal edin. Her anahtar, tam erişime veya salt okunur izinlere sahip belirli modellerle/sağlayıcılarla sınırlandırılabilir. Kullanım takibi ile görsel anahtar yönetimi.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Eylem türüne, aktöre, hedefe, IP adresine ve zaman damgasına göre filtrelemeyle idari işlem takibi. Tam güvenlik olayı geçmişi.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Windows, macOS ve Linux için Native Electron masaüstü uygulaması. OmniRoute'u sistem tepsisi entegrasyonu, çevrimdışı destek, otomatik güncelleme ve tek tıklamayla kurulum özellikleriyle bağımsız bir uygulama olarak çalıştırın. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Temel özellikler: +Key features: -- Sunucu hazırlığı yoklaması (soğuk başlangıçta boş ekran yok) -- Bağlantı noktası yönetimine sahip sistem tepsisi -- İçerik Güvenliği Politikası -- Tek örnekli kilit -- Yeniden başlatıldığında otomatik güncelleme -- Platform koşullu kullanıcı arayüzü (macOS trafik ışıkları, Windows/Linux varsayılan başlık çubuğu) -- Sertleştirilmiş Elektron yapı paketlemesi — bağımsız paketteki sembolik bağlantılı "node_modules" paketlemeden önce algılanır ve reddedilir, böylece çalışma zamanının yapı makinesine bağımlılığı önlenir (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Belgelerin tamamı için [`electron/README.md`](../electron/README.md) adresine bakın. +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/tr/docs/TROUBLESHOOTING.md b/docs/i18n/tr/docs/TROUBLESHOOTING.md index 5fbe740efb..49461d553e 100644 --- a/docs/i18n/tr/docs/TROUBLESHOOTING.md +++ b/docs/i18n/tr/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -OmniRoute için yaygın sorunlar ve çözümler.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Sorun | Çözüm | -| -------------------------------------------------- | ---------------------------------------------------------------------------------- | --- | -| İlk giriş çalışmıyor | `.env`de `INITIAL_PASSWORD`u ayarlayın (sabit kodlanmış varsayılan yok) | -| Kontrol paneli yanlış bağlantı noktasında açılıyor | `PORT=20128` ve `NEXT_PUBLIC_BASE_URL=http://localhost:20128` ayarını yapın | -| 'logs/' altında istek günlüğü yok | 'ENABLE_REQUEST_LOGS=true' olarak ayarlayın | -| EACCES: izin reddedildi | `~/.omniroute` geçersiz kılmak için `DATA_DIR=/path/to/writable/dir` ayarını yapın | -| Yönlendirme stratejisi kaydedilmiyor | v1.4.11+ Güncellemesi (Ayarların kalıcılığı için Zod şeması düzeltmesi) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Neden:**Sağlayıcı kotası doldu. +**Cause:** Provider quota exhausted. -**Düzeltme:** +**Fix:** -1. Kontrol paneli kota izleyicisini kontrol edin -2. Geri dönüş katmanlarına sahip bir kombinasyon kullanın -3. Daha ucuz/ücretsiz seviyeye geçin### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Neden:**Abonelik kotası tükendi. +### Rate Limiting -**Düzeltme:** +**Cause:** Subscription quota exhausted. -- Geri dönüş ekleyin: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- GLM/MiniMax'ı ucuz yedekleme olarak kullanın### OAuth Token Expired +**Fix:** -OmniRoute belirteçleri otomatik olarak yeniler. Sorunlar devam ederse: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Kontrol Paneli → Sağlayıcı → Yeniden Bağlan -2. Sağlayıcı bağlantısını silin ve yeniden ekleyin--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. "BASE_URL"nin çalışan örneğinize işaret ettiğini doğrulayın (ör. "http://localhost:20128") -2. "CLOUD_URL"nin bulut uç noktanıza işaret ettiğini doğrulayın (ör. "https://omniroute.dev") -3. `NEXT_PUBLIC_*` değerlerini sunucu tarafı değerleriyle uyumlu tutun### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Belirti:**Akış dışı aramalar için bulut uç noktasında "Beklenmeyen belirteç 'd'...'. +### Cloud `stream=false` Returns 500 -**Neden:**İstemci JSON beklerken yukarı akış SSE yükünü döndürüyor. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Geçici çözüm:**Buluttan doğrudan çağrılar için "stream=true" seçeneğini kullanın. Yerel çalışma zamanı SSE→JSON geri dönüşünü içerir.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Yerel kontrol panelinden yeni bir anahtar oluşturun (`/api/keys`) -2. Bulut senkronizasyonunu çalıştırın: Bulutu Etkinleştir → Şimdi Senkronize Et -3. Eski/senkronize edilmemiş anahtarlar bulutta hâlâ '401'i döndürebilir--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Çalışma zamanı alanlarını kontrol edin: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. Taşınabilir mod için: "runner-cli" görüntü hedefini kullanın (birlikte verilen CLI'ler) -3. Ana bilgisayar bağlama modu için: `CLI_EXTRA_PATHS`yi ayarlayın ve ana bilgisayar bin dizinini salt okunur olarak bağlayın -4. "Kurulu=doğru" ve "çalıştırılabilir=yanlış" ise: ikili dosya bulundu ancak durum denetimi başarısız oldu### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Kontrol Paneli → Kullanım bölümünden kullanım istatistiklerini kontrol edin -2. Birincil modeli GLM/MiniMax'a değiştirin -3. Kritik olmayan görevler için ücretsiz kullanımı (Gemini CLI, Qoder) kullanın -4. API anahtarı başına maliyet bütçelerini ayarlayın: Kontrol Paneli → API Anahtarları → Bütçe--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -`.env` dosyanızda `ENABLE_REQUEST_LOGS=true` değerini ayarlayın. Günlükler 'logs/' dizini altında görünür.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,96 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Ana durum: `${DATA_DIR}/storage.sqlite` (sağlayıcılar, kombinasyonlar, takma adlar, anahtarlar, ayarlar) -- Kullanım: `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) içindeki SQLite tabloları + isteğe bağlı `${DATA_DIR}/log.txt` ve `${DATA_DIR}/call_logs/` -- İstek günlükleri: `/logs/...` (`ENABLE_REQUEST_LOGS=true` olduğunda)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Sağlayıcının devre kesicisi AÇIK olduğunda, bekleme süresi dolana kadar istekler engellenir. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Düzeltme:** +**Fix:** -1.**Kontrol Paneli → Ayarlar → Dayanıklılık**'a gidin 2. Etkilenen sağlayıcının devre kesici kartını kontrol edin 3. Tüm kesicileri temizlemek için**Tümünü Sıfırla**'ya tıklayın veya bekleme süresinin dolmasını bekleyin 4. Sıfırlamadan önce sağlayıcının gerçekten kullanılabilir durumda olduğunu doğrulayın### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Bir sağlayıcı sürekli olarak AÇIK durumuna girerse: +### Provider keeps tripping the circuit breaker -1. Arıza modeli için**Kontrol Paneli → Sağlık → Sağlayıcı Sağlığı**'nı kontrol edin 2.**Ayarlar → Dayanıklılık → Sağlayıcı Profilleri**'ne gidin ve hata eşiğini artırın -2. Sağlayıcının API sınırlarını değiştirip değiştirmediğini veya yeniden kimlik doğrulama gerektirip gerektirmediğini kontrol edin -3. Gecikme telemetrisini gözden geçirin — yüksek gecikme, zaman aşımı temelli hatalara neden olabilir--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Doğru öneki kullandığınızdan emin olun: `deepgram/nova-3` veya `assemblyai/best` -- Sağlayıcının**Kontrol Paneli → Sağlayıcılar**'a bağlı olduğunu doğrulayın### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Desteklenen ses formatlarını kontrol edin: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- Dosya boyutunun sağlayıcı sınırları dahilinde olduğunu doğrulayın (genellikle < 25MB) -- Sağlayıcı kartındaki sağlayıcı API anahtarının geçerliliğini kontrol edin--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Biçim çeviri sorunlarının hatalarını ayıklamak için**Kontrol Paneli → Çevirmen**'i kullanın: +Use **Dashboard → Translator** to debug format translation issues: -| Modu | Ne Zaman Kullanılmalı | -| --------------------- | --------------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Oyun Alanı** | Giriş/çıkış formatlarını yan yana karşılaştırın; nasıl çevrildiğini görmek için başarısız bir isteği yapıştırın | -| **Sohbet Test Aracı** | Canlı mesajlar gönderin ve başlıklar dahil tüm istek/yanıt yükünü inceleyin | -| **Test Tezgahı** | Hangi çevirilerin bozuk olduğunu bulmak için format kombinasyonlarında toplu testler çalıştırın | -| **Canlı Monitör** | Aralıklı çeviri sorunlarını yakalamak için gerçek zamanlı istek akışını izleyin | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Düşünme etiketleri görünmüyor**— Hedef sağlayıcının düşünmeyi ve düşünme bütçesi ayarını destekleyip desteklemediğini kontrol edin -**Araç çağrıları bırakılıyor**— Bazı biçim çevirileri desteklenmeyen alanları kaldırabilir; Oyun Alanı modunda doğrula -**Sistem istemi eksik**— Claude ve Gemini sistemi istemleri farklı şekilde yönetir; çeviri çıktısını kontrol et -**SDK, nesne yerine ham dize döndürür**— V1.1.0'da düzeltildi: yanıt temizleyici artık OpenAI SDK Pydantic doğrulama hatalarına neden olan standart olmayan alanları ('x_groq', 'usage_breakdown' vb.) kaldırıyor -**GLM/ERNIE "sistem" rolünü reddediyor**— v1.1.0'da düzeltildi: rol normalleştirici, uyumsuz modeller için sistem mesajlarını otomatik olarak kullanıcı mesajlarıyla birleştiriyor -**'geliştirici' rolü tanınmıyor**- v1.1.0'da düzeltildi: OpenAI olmayan sağlayıcılar için otomatik olarak 'sistem'e dönüştürüldü -**`json_schema` Gemini ile çalışmıyor**— v1.1.0'da düzeltildi: `response_format` artık Gemini'nin `responseMimeType` + `responseSchema` biçimine dönüştürüldü--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- Otomatik hız sınırı yalnızca API anahtarı sağlayıcıları için geçerlidir (OAuth/abonelik için geçerli değildir) -**Ayarlar → Dayanıklılık → Sağlayıcı Profilleri**'nde otomatik hız sınırının etkin olduğunu doğrulayın -- Sağlayıcının '429' durum kodlarını mı yoksa 'Sonra Yeniden Dene' başlıklarını mı döndürdüğünü kontrol edin### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -Sağlayıcı profilleri şu ayarları destekler: +### Tuning exponential backoff --**Temel gecikme**— İlk arızadan sonraki ilk bekleme süresi (varsayılan: 1 saniye) -**Maksimum gecikme**— Maksimum bekleme süresi sınırı (varsayılan: 30 sn) -**Çarpan**— Ardışık arıza başına gecikmenin ne kadar artırılacağı (varsayılan: 2x)### Anti-thundering herd +Provider profiles support these settings: -Çok sayıda eşzamanlı istek, hızı sınırlı bir sağlayıcıya ulaştığında, OmniRoute, istekleri serileştirmek ve basamaklı hataları önlemek için mutex + otomatik hız sınırlamayı kullanır. Bu, API anahtarı sağlayıcıları için otomatiktir.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Bazı OmniRoute kullanıcıları ağ geçidini RAG veya aracı yığınlarının önüne yerleştirir. Bu kurulumlarda garip bir model görmek yaygındır: OmniRoute sağlıklı görünüyor (sağlayıcılar çalışıyor, yönlendirme profilleri iyi, hız sınırı uyarısı yok) ancak son yanıt hâlâ yanlış. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -Uygulamada bu olaylar genellikle ağ geçidinin kendisinden değil, aşağı yöndeki RAG boru hattından kaynaklanır. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Bu arızaları açıklamak için ortak bir kelime dağarcığı istiyorsanız, on altı yinelenen RAG / LLM arıza modelini tanımlayan harici bir MIT lisans metin kaynağı olan WFGY ProblemMap'i kullanabilirsiniz. Yüksek düzeyde şunları kapsar: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- sürüklenmeyi ve bozulmuş bağlam sınırlarını geri getirme -- boş veya eski dizinler ve vektör depoları -- anlamsal uyumsuzluğa karşı yerleştirme -- hızlı derleme ve bağlam penceresi sorunları -- Mantık çöküşü ve kendine aşırı güvenen cevaplar -- uzun zincir ve temsilci koordinasyon hataları -- çoklu ajan hafızası ve rol kayması -- dağıtım ve önyükleme sıralama sorunları +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -Fikir basit: +The idea is simple: -1. Kötü bir yanıtı araştırırken şunları yakalayın: - - kullanıcı görevi ve isteği - - OmniRoute'ta rota veya sağlayıcı birleşimi - - aşağı yönde kullanılan herhangi bir RAG bağlamı (alınan belgeler, araç çağrıları vb.) +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) 2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). -3. Numarayı kendi kontrol panelinizde, runbook'unuzda veya olay izleyicinizde OmniRoute günlüklerinin yanında saklayın. -4. RAG yığınınızı, alıcınızı veya yönlendirme stratejinizi değiştirmeniz gerekip gerekmediğine karar vermek için ilgili WFGY sayfasını kullanın. +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -Tam metin ve somut tarifler burada yayınlanmaktadır (MIT lisansı, yalnızca metin): +Full text and concrete recipes live here (MIT license, text only): -[WFGY ProblemMap BENİ OKU](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) +[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -OmniRoute'un arkasında RAG veya aracı işlem hatlarını çalıştırmıyorsanız bu bölümü göz ardı edebilirsiniz.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**GitHub Sorunları**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Mimarlık**: Dahili ayrıntılar için bkz. [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) -**API Referansı**: Tüm uç noktalar için bkz. [`docs/API_REFERENCE.md`](API_REFERENCE.md) -**Sağlık Kontrol Paneli**: Gerçek zamanlı sistem durumu için**Kontrol Paneli → Sağlık**'ı kontrol edin -**Çevirmen**: Biçim sorunlarının hatalarını ayıklamak için**Kontrol Paneli → Çevirmen**'i kullanın +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/tr/llm.txt b/docs/i18n/tr/llm.txt new file mode 100644 index 0000000000..afa33a9e19 --- /dev/null +++ b/docs/i18n/tr/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Türkçe) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Genel Bakış + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Güvenlik +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/uk-UA/README.md b/docs/i18n/uk-UA/README.md index f60a58f0cb..9b43547a34 100644 --- a/docs/i18n/uk-UA/README.md +++ b/docs/i18n/uk-UA/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Ваш універсальний API-проксі — одна кінцева точка, понад 60 провайдерів, нуль простоїв. Тепер із**MCP Server (25 інструментів)**,**A2A Protocol**,**Memory/Skills Systems**і**Electron Desktop App**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Завершення чату • Вбудовування • Генерація зображень • Відео • Музика • Аудіо • Переранжування •**Веб-пошук**• Сервер MCP • Протокол A2A • 100% TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Ваш універсальний API-проксі — одна кінцева [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Веб-сайт](https://omniroute.online) • [🚀 Швидкий старт](#-швидкий-старт) • [💡 Функції](#-key-features) • [📖 Документи](#-документація) • [💰 Ціни](#-pricing-at-lance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Доступно:**🇺🇸 [англійською](README.md) | 🇧🇷 [Португальська (Бразилія)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Італійська](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Угорська](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Нідерланди](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Португальська (Португалія)](docs/i18n/pt/README.md) | 🇷🇴 [Руманська](docs/i18n/ro/README.md) | 🇵🇱 [Польський](docs/i18n/pl/README.md) | 🇸🇰 [Словенчина](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [філіппінський](docs/i18n/phi/README.md) | 🇨🇿 [Чештина](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,555 +60,629 @@ _Ваш універсальний API-проксі — одна кінцева ## 📸 Dashboard Preview -<подробиці> +
+Click to see dashboard screenshots -Натисніть, щоб переглянути знімки екрана інформаційної панелі +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | -| Сторінка | Скріншот | -| ------------------------ | ----------------------------------------------------- | ---------- | -| **Постачальники** | ![Постачальники](docs/screenshots/01-providers.png) | -| **Комбінації** | ![Комбінації](docs/screenshots/02-combos.png) | -| **Аналітика** | ![Аналітика](docs/screenshots/03-analytics.png) | -| **Здоров'я** | ![Health](docs/screenshots/04-health.png) | -| **Перекладач** | ![Перекладач](docs/screenshots/05-translator.png) | -| **Налаштування** | ![Налаштування](docs/screenshots/06-settings.png) | -| **Інструменти CLI** | ![Інструменти CLI](docs/screenshots/07-cli-tools.png) | -| **Журнали використання** | ![Використання](docs/screenshots/08-usage.png) | -| **Кінцеві точки** | ![Кінцеві точки](docs/screenshots/09-endpoint.png) |
| + --- ### 🤖 Free AI Provider for your favorite coding agents -_Підключіть будь-який інструмент IDE або CLI на основі штучного інтелекту через OmniRoute — безкоштовний шлюз API для необмеженого програмування._ - -<таблиця> - - - -OpenClaw
-OpenClaw -

-⭐ 205K - - - -NanoBot
-NanoBot -

-⭐ 20,9 тис. - - - -PicoClaw
-PicoClaw -

-⭐ 14,6 тис. - - - -ZeroClaw
-ZeroClaw -

-⭐ 9,9 тис. - - - -IronClaw
-Залізний Кіготь -

-⭐ 2,1 тис. - - - - - -OpenCode
-OpenCode -

-⭐ 106K - - - -Codex CLI
-Codex CLI -

-⭐ 60,8 тис. - - - -Claude Code
-Клод Код -

-⭐ 67,3 тис. - - - -Gemini CLI
-Gemini CLI -

-⭐ 94,7 тис. - - - -Kilo Code
-Код Кіло -

-⭐ 15,5 тис. - - +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ + + + + + + + + + + + + + + +
+ + OpenClaw
+ OpenClaw +

+ ⭐ 205K +
+ + NanoBot
+ NanoBot +

+ ⭐ 20.9K +
+ + PicoClaw
+ PicoClaw +

+ ⭐ 14.6K +
+ + ZeroClaw
+ ZeroClaw +

+ ⭐ 9.9K +
+ + IronClaw
+ IronClaw +

+ ⭐ 2.1K +
+ + OpenCode
+ OpenCode +

+ ⭐ 106K +
+ + Codex CLI
+ Codex CLI +

+ ⭐ 60.8K +
+ + Claude Code
+ Claude Code +

+ ⭐ 67.3K +
+ + Gemini CLI
+ Gemini CLI +

+ ⭐ 94.7K +
+ + Kilo Code
+ Kilo Code +

+ ⭐ 15.5K +
-📡 Усі агенти підключаються через http://localhost:20128/v1 або http://cloud.omniroute.online/v1 — одна конфігурація, необмежена кількість моделей і квот--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Припиніть витрачати гроші та досягати лімітів:** +**Stop wasting money and hitting limits:** -- Квота підписки закінчується без використання кожного місяця -- Обмеження швидкості перешкоджають кодуванню в середині -- Дорогі API ($20-50/місяць за постачальника) -- Ручне перемикання між провайдерами +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute вирішує цю проблему:** +**OmniRoute solves this:** -- ✅**Збільште кількість підписок**- Відстежуйте квоту, використовуйте кожен біт перед скиданням -- ✅**Автоматичний резерв**- Підписка → Ключ API → Дешево → Безкоштовно, без простоїв -- ✅**Кілька облікових записів**- Циклічний цикл між обліковими записами кожного постачальника -- ✅**Універсальний**- Працює з Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, будь-яким інструментом CLI--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Приєднуйтесь до нашої спільноти!**[WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — отримуйте допомогу, діліться порадами та залишайтеся в курсі подій. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Веб-сайт**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Проблеми**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Група спільноти](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Допомога**: перегляньте [CONTRIBUTING.md](CONTRIBUTING.md), відкрийте PR або виберіть «хороший перший номер» -**Оригінальний проект**: [9router від decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Відкриваючи проблему, виконайте команду system-info та прикріпіть згенерований файл:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Це генерує `system-info.txt` з вашою версією Node.js, версією OmniRoute, деталями ОС, встановленими інструментами CLI (qoder, gemini, claude, codex, antigravity, droid тощо), статусом Docker/PM2 і системними пакетами — усім, що нам потрібно для швидкого відтворення вашої проблеми. Прикріпіть файл безпосередньо до свого випуску GitHub.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Кожен розробник, який використовує інструменти штучного інтелекту, щодня стикається з цими проблемами.**OmniRoute було створено, щоб вирішити їх усі — від перевитрати коштів до регіональних блокувань, від порушених потоків OAuth до операцій протоколу та можливості спостереження на підприємстві. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. -<подробиці> -💸 1. «Я плачу за дорогу підписку, але мені все одно переривається через обмеження» +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Розробники платять 20–200 доларів на місяць за Claude Pro, Codex Pro або GitHub Copilot. Навіть якщо платити, квота має максимальну межу — 5 годин використання, тижневі ліміти або ліміти за хвилину. У середині сеансу кодування постачальник перестає відповідати, а розробник втрачає швидкість і продуктивність. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Як це вирішує OmniRoute:** +**How OmniRoute solves it:** --**Smart 4-Tier Fallback**— якщо квота підписки закінчується, автоматично перенаправляється до API Key → Дешево → Безкоштовно без ручного втручання --**Відстеження обмежень постачальника**— кешовані знімки квот оновлюються за розкладом на стороні сервера (за замовчуванням `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) з ручним оновленням, доступним в інтерфейсі користувача. --**Підтримка кількох облікових записів**— кілька облікових записів у кожного постачальника з автоматичним циклічним перебором — коли один закінчується, перемикається на наступний --**Користувацькі комбінації**— резервні ланцюжки, що налаштовуються, із 9 стратегіями балансування (пріоритетні, зважені, спочатку заповнюють, циклічні, P2C, випадкові, найменш використовувані, оптимізовані за витратами, строго випадкові) --**Бізнес-квоти Codex**— Моніторинг квот робочого простору бізнесу/команди безпосередньо на інформаційній панелі
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard -<подробиці> -🔌 2. "Мені потрібно використовувати кілька постачальників, але кожен має інший API" + -OpenAI використовує один формат, Claude (Anthropic) використовує інший, Gemini ще інший. Якщо розробник хоче перевірити моделі від різних постачальників або повернутися до них, йому потрібно переналаштувати SDK, змінити кінцеві точки, мати справу з несумісними форматами. Спеціальні постачальники (FriendLI, NIM) мають нестандартні кінцеві точки моделі. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Як це вирішує OmniRoute:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Unified Endpoint**— єдиний `http://localhost:20128/v1` служить проксі-сервером для всіх 60+ провайдерів --**Переклад форматів**— автоматичний і прозорий: OpenAI ↔ Claude ↔ Gemini ↔ Responses API --**Response Sanitization**— видаляє нестандартні поля (`x_groq`, `usage_breakdown`, `service_tier`), які порушують роботу OpenAI SDK v1.83+ --**Нормалізація ролі**— перетворює `розробник` в `систему` для постачальників, які не є OpenAI; `система` → `користувач` для GLM/ERNIE --**Think Tag Extraction**— витягує блоки з таких моделей, як DeepSeek R1, у стандартизований `reasoning_content` --**Структурований вихід для Gemini**— `json_schema` → `responseMimeType`/`responseSchema` автоматичне перетворення --**`потік` за умовчанням має значення `false`**— узгоджується зі специфікацією OpenAI, уникаючи неочікуваних SSE у Python/Rust/Go SDK.
+**How OmniRoute solves it:** -<подробиці> -🌐 3. «Мій постачальник штучного інтелекту блокує мій регіон/країну» +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Такі постачальники, як OpenAI/Codex, блокують доступ із певних географічних регіонів. Користувачі отримують помилки на зразок `unsupported_country_region_territory` під час підключень OAuth і API. Це особливо засмучує розробників із країн, що розвиваються. + -**Як це вирішує OmniRoute:** +
+🌐 3. "My AI provider blocks my region/country" --**3-рівнева конфігурація проксі-сервера**— налаштовується проксі-сервер на 3 рівнях: глобальний (увесь трафік), для кожного постачальника (лише один постачальник) і для кожного підключення/ключа --**Значки проксі-сервера з кольоровим кодуванням**— Візуальні індикатори: 🟢 глобальний проксі, 🟡 проксі-сервер постачальника, 🔵 проксі-сервер підключення, завжди показує IP-адресу --**Обмін маркерами OAuth через проксі**— потік OAuth також проходить через проксі, вирішуючи `unsupported_country_region_territory` --**Тестування з’єднання через проксі**— тестування з’єднання використовує налаштований проксі (без прямого обходу) --**Підтримка SOCKS5**— повна підтримка проксі SOCKS5 для вихідної маршрутизації --**TLS Fingerprint Spoofing**— відбиток TLS, подібний до браузера, через `wreq-js` для обходу виявлення ботів --**🔏 Зіставлення відбитків CLI**— змінює порядок заголовків і полів основного вмісту, щоб відповідати власним двійковим підписам CLI, суттєво знижуючи ризик позначення облікового запису. IP-адреса проксі-сервера зберігається — ви отримуєте приховане**і**маскування IP-адреси одночасно
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. -<подробиці> -🆓 4. «Я хочу використовувати ШІ для кодування, але в мене немає грошей» +**How OmniRoute solves it:** -Не кожен може платити 20–200 доларів на місяць за підписку на AI. Студентам, розробникам із країн, що розвиваються, любителям і фрілансерам потрібен доступ до якісних моделей за нульовою ціною. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Як це вирішує OmniRoute:** + --**Вбудовані безкоштовні постачальники рівня**— Вбудована підтримка 100% безкоштовних постачальників: Qoder (5 необмежених моделей через OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 необмежені моделі: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID безкоштовно), Gemini CLI (180K токенів/місяць безкоштовно) --**Ollama Cloud**— моделі Ollama, розміщені в хмарі на `api.ollama.com` з безкоштовним рівнем "Light usage"; використовуйте префікс `ollamacloud/<модель>` --**Тільки безкоштовні комбінації**— Ланцюжок `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = 0 доларів США/місяць без простою --**Безкоштовний доступ до NVIDIA NIM**— ~40 об./хв. для розробників назавжди безкоштовний доступ до 70+ моделей на build.nvidia.com (перехід від кредитів до чистих обмежень швидкості) --**Стратегія оптимізації витрат**— стратегія маршрутизації, яка автоматично вибирає найдешевшого доступного постачальника +
+🆓 4. "I want to use AI for coding but I have no money" -<подробиці> -🔒 5. «Мені потрібно захистити мій штучний шлюз від несанкціонованого доступу» +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -Коли шлюз штучного інтелекту надається мережі (LAN, VPS, Docker), будь-хто, хто має адресу, може використовувати токени/квоту розробника. Без захисту API вразливі до неправильного використання, швидкого впровадження та зловживання. +**How OmniRoute solves it:** -**Як це вирішує OmniRoute:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**Керування ключами API**— генерація, ротація та визначення обсягу для кожного постачальника за допомогою спеціальної сторінки `/dashboard/api-manager` --**Дозволи на рівні моделі**— обмежте ключі API певними моделями (`openai/*`, шаблони підстановки), з перемикачем Дозволити все/Обмежити --**API Endpoint Protection**— вимагати ключ для `/v1/models` і блокувати певних постачальників зі списку --**Auth Guard + CSRF Protection**— усі маршрути інформаційної панелі захищені проміжним програмним забезпеченням `withAuth` + токени CSRF --**Обмежувач швидкості**— обмеження швидкості за IP-адресою з настроюваними вікнами --**IP Filtering**— список дозволених/чорних адрес для контролю доступу --**Prompt Injection Guard**— очищення від шкідливих шаблонів підказок --**Шифрування AES-256-GCM**— облікові дані зашифровані в стані спокою
+ -<подробиці> -🛑 6. «Мій постачальник не працює, і я втратив потік кодування» +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -Постачальники AI можуть стати нестабільними, повертати помилки 5xx або досягати тимчасових обмежень швидкості. Якщо розробник залежить від одного постачальника, вони перериваються. Без автоматичних вимикачів повторні спроби можуть призвести до збою програми. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Як це вирішує OmniRoute:** +**How OmniRoute solves it:** --**Автоматичний вимикач для кожної моделі**— Автоматичне розмикання/замикання з настроюваними пороговими значеннями та часом охолодження (замкнуто/розімкнуто/напіврозімкнуто), для кожної моделі, щоб уникнути каскадних блокувань --**Exponential Backoff**— прогресивні затримки повторних спроб --**Anti-Thundering Herd**— Mutex + захист семафора від одночасних повторних штормів --**Комбіновані запасні ланцюги**— якщо основний постачальник виходить з ладу, автоматично проходить через ланцюжок без втручання. --**Combo Circuit Breaker**— автоматично вимикає несправні постачальники в комбінованому ланцюжку --**Health Dashboard**— Моніторинг безвідмовної роботи, стани автоматичного вимикача, блокування, статистика кешу, затримка p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest -<подробиці> -🔧 7. «Налаштування кожного інструменту штучного інтелекту – справа втомлива та повторювана» + -Розробники використовують Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Кожному інструменту потрібна інша конфігурація (кінцева точка API, ключ, модель). Перенастроювання при зміні постачальника чи моделі – марна трата часу. +
+🛑 6. "My provider went down and I lost my coding flow" -**Як це вирішує OmniRoute:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**Панель інструментів CLI**— спеціальна сторінка з налаштуванням одним клацанням для Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline --**GitHub Copilot Config Generator**— генерує `chatLanguageModels.json` для коду VS із масовим вибором моделі --**Майстер адаптації**— 4-етапне налаштування для тих, хто вперше користується --**Одна кінцева точка, усі моделі**— налаштуйте `http://localhost:20128/v1` один раз, отримайте доступ до 60+ постачальників
+**How OmniRoute solves it:** -<подробиці> -🔑 8. «Керування маркерами OAuth від кількох постачальників — це пекло» +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot — усі використовують OAuth 2.0 із терміном дії маркерів. Розробникам потрібно постійно проходити повторну автентифікацію, мати справу з `client_secret is missing`, `redirect_uri_mismatch` та збоями на віддалених серверах. OAuth у LAN/VPS є особливо проблематичним. + -**Як це вирішує OmniRoute:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Auto Token Refresh**— маркери OAuth оновлюються у фоновому режимі до завершення терміну дії --**Вбудований OAuth 2.0 (PKCE)**— автоматичний потік для Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder --**Multi-Account OAuth**— кілька облікових записів на постачальника за допомогою вилучення токенів JWT/ID --**OAuth LAN/Remote Fix**— виявлення приватної IP-адреси для `redirect_uri` + ручний режим URL-адреси для віддалених серверів --**OAuth за Nginx**— використовує `window.location.origin` для зворотної сумісності проксі --**Remote OAuth Guide**— покроковий посібник для облікових даних Google Cloud на VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. -<подробиці> -📊 9. «Я не знаю, скільки я витрачаю і куди» +**How OmniRoute solves it:** -Розробники використовують кілька платних постачальників, але не мають єдиного уявлення про витрати. Кожен постачальник має власну платіжну панель, але немає консолідованого перегляду. Несподівані витрати можуть накопичитися. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Як це вирішує OmniRoute:** + --**Інформаційна панель аналітики витрат**— відстеження вартості кожного токена та керування бюджетом для кожного постачальника --**Бюджетні обмеження на рівень**— максимальна сума витрат на рівень, що запускає автоматичний відкат --**Конфігурація цін за моделлю**— настроювані ціни за модель --**Статистика використання за ключ API**— кількість запитів і позначка часу останнього використання для кожного ключа --**Інформаційна панель аналітики**— статистичні картки, діаграма використання моделі, таблиця постачальників із показниками успішності та затримкою +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" -<подробиці> -🐛 10. «Я не можу діагностувати помилки та проблеми під час викликів ШІ» +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Коли виклик не вдається, розробник не знає, чи це було обмеження швидкості, прострочений маркер, неправильний формат чи помилка постачальника. Фрагментовані журнали на різних терміналах. Без спостережливості налагодження відбувається методом проб і помилок. +**How OmniRoute solves it:** -**Як це вирішує OmniRoute:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Інформаційна панель уніфікованих журналів**— 4 вкладки: журнали запитів, журнали проксі, журнали аудиту, консоль --**Console Log Viewer**— засіб перегляду терміналів у режимі реального часу з кольоровими рівнями, автопрокручуванням, пошуком, фільтром --**Проксі-журнали SQLite**— постійні журнали, які залишаються після перезапуску сервера --**Translator Playground**— 4 режими налагодження: Playground (переклад формату), Chat Tester (туди й назад), Test Bench (пакет), Live Monitor (у реальному часі) --**Запит телеметрії**— затримка p50/p95/p99 + трасування X-Request-Id --**Файлове журналювання з ротацією**— журнали програми чергуються за розміром, днями зберігання та кількістю архівів; артефакти журналу викликів чергуються за днями зберігання та кількістю файлів --**Звіт про інформацію про систему**— `npm run system-info` генерує `system-info.txt` із вашим повним середовищем (версія Node, версія OmniRoute, ОС, інструменти CLI, статус Docker/PM2). Додайте його під час повідомлення про проблеми для миттєвого сортування.
+ -<подробиці> -🏗️ 11. «Розгортати та підтримувати шлюз складно» +
+📊 9. "I don't know how much I'm spending or where" -Встановлення, налаштування та обслуговування проксі ШІ в різних середовищах (локальне, VPS, Docker, хмара) є трудомістким. Такі проблеми, як жорстко закодовані шляхи, `EACCES` у каталогах, конфлікти портів і кросплатформні збірки, додають тертя. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Як це вирішує OmniRoute:** +**How OmniRoute solves it:** --**npm global install**— `npm install -g omniroute && omniroute` — готово --**Мультиплатформенний Docker**— нативний AMD64 + ARM64 (Apple Silicon, AWS Graviton, Raspberry Pi) --**Docker Compose Profiles**— `base` (без інструментів CLI) і `cli` (з Claude Code, Codex, OpenClaw) --**Electron Desktop App**— рідна програма для Windows/macOS/Linux із системним треєм, автозапуском, офлайн-режимом --**Режим розділеного порту**— API та інформаційна панель на окремих портах для розширених сценаріїв (зворотний проксі, мережа контейнерів) --**Cloud Sync**— синхронізація налаштувань між пристроями через Cloudflare Workers --**Резервне копіювання БД**— автоматичне резервне копіювання, відновлення, експорт і імпорт усіх налаштувань із `DISABLE_SQLITE_AUTO_BACKUP` для зовнішніх керованих резервних копій
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency -<подробиці> -🌍 12. «Інтерфейс лише англійською мовою, і моя команда не розмовляє англійською» + -Команди в неангломовних країнах, особливо в Латинській Америці, Азії та Європі, стикаються з інтерфейсами лише англійською мовою. Мовні бар’єри зменшують адаптацію та збільшують кількість помилок конфігурації. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Як це вирішує OmniRoute:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Інформаційна панель i18n — 30 мов**— перекладено всі 500+ клавіш, включаючи арабську, болгарську, датську, німецьку, іспанську, фінську, французьку, іврит, гінді, угорську, індонезійську, італійську, японську, корейську, малайську, голландську, норвезьку, польську, португальську (PT/BR), румунську, російську, словацьку, шведську, тайську, українську, в’єтнамську, китайська, філіппінська, англійська --**Підтримка RTL**— підтримка арабської та івриту справа наліво --**Багатомовні файли README**— 30 повних перекладів документації --**Вибір мови**— значок глобуса в заголовку для перемикання в реальному часі
+**How OmniRoute solves it:** -<подробиці> -🔄 13. «Мені потрібно більше, ніж чат — мені потрібні вставки, зображення, аудіо» +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -ШІ — це не просто завершення чату. Розробникам потрібно генерувати зображення, транскрибувати аудіо, створювати вбудовування для RAG, змінювати рейтинг документів і модерувати вміст. Кожен API має різну кінцеву точку та формат. + -**Як це вирішує OmniRoute:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Вбудовування**— `/v1/вбудовування` з 6 постачальниками та 9+ моделями --**Генерація зображень**— `/v1/images/generations` з 10 постачальниками та понад 20 моделями (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) --**Текст у відео**— `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) і SD WebUI --**Текст у музику**— `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) --**Транскрипція аудіо**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Перетворення тексту в мовлення**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + існуючі постачальники --**Модерації**— `/v1/moderations` — Перевірка безпеки вмісту --**Переранжування**— `/v1/rerank` — Переранжування релевантності документа --**Responses API**— повна підтримка `/v1/responses` для Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. -<подробиці> -🧪 14. «У мене немає можливості перевірити та порівняти якість різних моделей» +**How OmniRoute solves it:** -Розробники хочуть знати, яка модель найкраще підходить для їхнього випадку використання — код, переклад, міркування — але порівнювати вручну повільно. Інтегрованих інструментів оцінювання не існує. +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**Як це вирішує OmniRoute:** + --**Оцінки LLM**— Золотий набір тестів із 10 попередньо завантаженими випадками, що охоплюють привітання, математику, географію, генерацію коду, відповідність JSON, переклад, уцінку, відмову безпеки --**4 стратегії відповідності**— `exact`, `contains`, `regex`, `custom` (функція JS) --**Translator Playground Test Bench**— Пакетне тестування з кількома входами та очікуваними результатами, порівняння між постачальниками --**Chat Tester**— повний цикл із візуальним відтворенням відповідей --**Live Monitor**— потік усіх запитів, що проходять через проксі, у реальному часі +
+🌍 12. "The interface is English-only and my team doesn't speak English" -<подробиці> -📈 15. «Мені потрібно масштабувати без втрати продуктивності» +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -Оскільки кількість запитів зростає, без кешування ті самі запитання створюють дублюючі витрати. Без ідемпотентності, дублікат запитів обробки відходів. Необхідно дотримуватися обмежень тарифів для кожного постачальника. +**How OmniRoute solves it:** -**Як це вирішує OmniRoute:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**Семантичний кеш**— дворівневий кеш (підпис + семантичний) зменшує вартість і затримку --**Request Idempotency**— вікно дедуплікації 5 с для ідентичних запитів --**Виявлення ліміту швидкості**— RPM для кожного постачальника, мінімальний розрив і максимальне одночасне відстеження --**Обмеження швидкості, які можна редагувати**— налаштування за замовчуванням у Параметрах → Стійкість із наполегливістю --**API Key Validation Cache**— 3-рівневий кеш для продуктивності --**Інформаційна панель справності з телеметрією**— затримка p50/p95/p99, статистика кешу, час роботи
+ -<подробиці> -🤖 16. «Я хочу глобально контролювати поведінку моделі» +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -Розробники, які хочуть, щоб усі відповіді відповідали певною мовою, з певним тоном або хочуть обмежити маркери міркування. Налаштовувати це в кожному інструменті/запиті є недоцільним. +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**Як це вирішує OmniRoute:** +**How OmniRoute solves it:** --**Впровадження системної підказки**— глобальна підказка застосовується до всіх запитів --**Thinking Budget Validation**— Контроль розподілу токенів міркувань за запитом (прохідний, автоматичний, спеціальний, адаптивний) --**9 стратегій маршрутизації**— глобальні стратегії, які визначають спосіб розподілу запитів --**Wildcard Router**— шаблони `provider/*` динамічно маршрутизують до будь-якого постачальника --**Увімкнути/вимкнути комбо**— перемикайте комбо безпосередньо з інформаційної панелі --**Перемикнути постачальника**— увімкнути/вимкнути всі з’єднання для постачальника одним клацанням миші --**Заблоковані постачальники**— виключити певних постачальників зі списку `/v1/models`
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex -<подробиці> -🧰 17. «Мені потрібні інструменти MCP як першокласні можливості продукту» + -Багато шлюзів ШІ розкривають MCP лише як приховану деталь реалізації. Командам потрібен видимий, керований рівень операцій. +
+🧪 14. "I have no way to test and compare quality across models" -**Як це вирішує OmniRoute:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP з’являється на панелі навігації та на вкладці протоколу кінцевої точки -- Спеціальна сторінка керування MCP із процесом, інструментами, обсягами й аудитом -- Вбудований швидкий запуск для `omniroute --mcp` і підключення клієнта
+**How OmniRoute solves it:** -<подробиці> -🧠 18. «Мені потрібна оркестровка A2A із синхронізацією + шляхи завдань потоку» +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -Робочі процеси агентів потребують як прямих відповідей, так і тривалого потокового виконання з контролем життєвого циклу. + -**Як це вирішує OmniRoute:** +
+📈 15. "I need to scale without losing performance" -- Кінцева точка A2A JSON-RPC (`POST /a2a`) з `message/send` і `message/stream` -- Потокова передача SSE з розповсюдженням стану терміналу -- API життєвого циклу завдань для `tasks/get` і `tasks/cancel`
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. -<подробиці> -🛰️ 19. «Мені потрібна реальна справність процесу MCP, а не вгаданий статус» +**How OmniRoute solves it:** -Операційним групам потрібно знати, чи справді MCP активний, а не лише те, чи доступний API. +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**Як це вирішує OmniRoute:** + -- Файл серцевого ритму виконання з PID, часовими мітками, транспортом, кількістю інструментів і режимом області -- API статусу MCP, що поєднує серцебиття + останню активність -- Картки стану інтерфейсу користувача для процесу/часу безперебійної роботи/свіжості пульсу +
+🤖 16. "I want to control model behavior globally" -<подробиці> -📋 20. «Мені потрібне виконання інструменту MCP з можливістю перевірки» +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -Коли інструменти змінюють конфігурацію або запускають операційні дії, командам потрібна криміналістична відстежуваність. +**How OmniRoute solves it:** -**Як це вирішує OmniRoute:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -— Журнал аудиту з підтримкою SQLite для викликів інструментів MCP -- Фільтри за інструментом, успіхом/невдачею, ключем API та розбивкою на сторінки -- Таблиця аудиту інформаційної панелі + кінцеві точки статистики для автоматизації
+ -<подробиці> -🔐 21. «Мені потрібні обмежені дозволи MCP для інтеграції» +
+🧰 17. "I need MCP tools as first-class product capabilities" -Різні клієнти повинні мати мінімальний доступ до категорій інструментів. +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**Як це вирішує OmniRoute:** +**How OmniRoute solves it:** -- 10 гранульованих областей MCP для контрольованого доступу до інструменту -- Застосування обсягу та видимість в інтерфейсі користувача керування MCP -- Безпечна поза за замовчуванням для робочих інструментів
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding -<подробиці> -⚙️ 22. «Мені потрібен оперативний контроль без передислокації» + -Командам потрібні швидкі зміни часу виконання під час інцидентів або витрат. +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**Як це вирішує OmniRoute:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- Перемикання комбо-активації безпосередньо з інформаційної панелі MCP -- Застосуйте профілі стійкості з попередньо визначених пакетів політик -- Скинути стан автоматичного вимикача з тієї ж панелі керування
+**How OmniRoute solves it:** -<подробиці> -🔄 23. "Мені потрібна видимість і скасування життєвого циклу завдання A2A" +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -Без видимості життєвого циклу інциденти завдань стає важко сортувати. + -**Як це вирішує OmniRoute:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -— Список завдань/фільтрування за станом/навиками з розбивкою на сторінки -- Деталізація метаданих завдань, подій і артефактів -- Кінцева точка скасування завдання та дія інтерфейсу користувача з підтвердженням
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. -<подробиці> -🌊 24. «Мені потрібні активні метрики потоку для завантаження A2A» +**How OmniRoute solves it:** -Робочі процеси потокової передачі вимагають оперативного розуміння паралельності та живих з’єднань. +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**Як це вирішує OmniRoute:** + -— Лічильники активних потоків інтегровані в статус A2A -- Мітка часу останнього завдання та підрахунок стану -- Картки інформаційної панелі A2A для моніторингу операцій у реальному часі +
+📋 20. "I need auditable MCP tool execution" -<подробиці> -🪪 25. «Мені потрібне стандартне виявлення агентів для клієнтів» +When tools mutate config or trigger ops actions, teams need forensic traceability. -Зовнішнім клієнтам і оркестрантам потрібні машинозчитувані метадані для адаптації. +**How OmniRoute solves it:** -**Як це вирішує OmniRoute:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- Картка агента розкрита в `/.well-known/agent.json` -- Можливості та навички, показані в інтерфейсі користувача користувача -- API стану A2A включає метадані виявлення для автоматизації
+ -<подробиці> -🧭 26. «Мені потрібна можливість виявлення протоколу в UX продукту» +
+🔐 21. "I need scoped MCP permissions per integration" -Якщо користувачі не можуть виявити поверхні протоколу, якість впровадження та підтримки падає. +Different clients should have least-privilege access to tool categories. -**Як це вирішує OmniRoute:** +**How OmniRoute solves it:** -- Консолідована сторінка**Кінцеві точки**з вкладками для кінцевих точок проксі, MCP, A2A та API -— Перемикання стану вбудованої служби (онлайн/офлайн) для MCP і A2A -- Посилання з огляду на спеціальні вкладки керування
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling -<подробиці> -🧪 27. «Мені потрібна наскрізна перевірка протоколу з реальними клієнтами» + -Пробних тестів недостатньо для перевірки сумісності протоколу перед випуском. +
+⚙️ 22. "I need operational controls without redeploying" -**Як це вирішує OmniRoute:** +Teams need quick runtime changes during incidents or cost events. -- Комплект E2E, який завантажує програму та використовує реальний клієнтський транспорт MCP SDK -- Клієнт A2A перевіряє потоки виявлення, надсилання, потокової передачі, отримання та скасування -- Перехресна перевірка тверджень щодо аудиту MCP та API завдань A2A
+**How OmniRoute solves it:** -<подробиці> -📡 28. «Мені потрібна уніфікована можливість спостереження через усі інтерфейси» +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -Поділ спостережуваності за протоколом створює сліпі зони та довший MTTR. + -**Як це вирішує OmniRoute:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- Уніфіковані інформаційні панелі/журнали/аналітика в одному продукті -- Справність + аудит + телеметрія запитів на рівнях OpenAI, MCP і A2A -- Операційні API для статусу та автоматизації
+Without lifecycle visibility, task incidents become hard to triage. -<подробиці> -💼 29. «Мені потрібен один час виконання для проксі + інструменти + оркестровка агента» +**How OmniRoute solves it:** -Запуск багатьох окремих служб збільшує експлуатаційні витрати та частоту збоїв. +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**Як це вирішує OmniRoute:** + -- OpenAI-сумісний проксі, сервер MCP і сервер A2A в одному стеку -- Спільна автентифікація, стійкість, зберігання даних і можливість спостереження -- Послідовна модель політики на всіх поверхнях взаємодії +
+🌊 24. "I need active stream metrics for A2A load" -<подробиці> -🚀 30. «Мені потрібно надсилати агентські робочі процеси без розповсюдження клею-коду» +Streaming workflows require operational insight into concurrency and live connections. -Команди втрачають швидкість під час з’єднання кількох спеціальних служб і сценаріїв. +**How OmniRoute solves it:** -**Як це вирішує OmniRoute:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- Уніфікована стратегія кінцевих точок для клієнтів і агентів -— Вбудовані інтерфейси керування протоколами та шляхи перевірки диму -- Основи, готові до виробництва (безпека, журналювання, стійкість, резервне копіювання)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Playbook A: Максимальна кількість платної підписки + дешеве резервне копіювання**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -609,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Playbook B: стек кодування без витрат**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Playbook C: резервний ланцюжок 24/7**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -632,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Playbook D: Операції агента з MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Налаштуйте кодування AI за лічені хвилини за**$0/місяць**. Підключіть ці безкоштовні облікові записи та використовуйте вбудовану комбінацію**Free Stack**. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Крок | Дія | Постачальники розблоковано | +| Step | Action | Providers Unlocked | | ---- | -------------------------------------------------- | ------------------------------------------------------------------ | -| 1 | Підключіть**Kiro**(AWS Builder ID OAuth) | Клод Сонет 4.5, Хайку 4.5 —**необмежений**| -| 2 | Підключіть**Qoder**(Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... —**необмежено**| -| 3 | Підключіть**Qwen**(код пристрою) | qwen3-coder-plus, qwen3-coder-flash... —**необмежено**| -| 4 | Підключіть**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180K/міс безкоштовно**| -| 5 | `/dashboard/combos` → Шаблон**Безкоштовний стек ($0)**| Циклічний цикл усіх безкоштовних постачальників автоматично | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Направте будь-який IDE/CLI на:**`http://localhost:20128/v1` · API-ключ: `any-string` · Готово. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Додаткове покриття (також безкоштовно):**ключ Groq API (30 об/хв безкоштовно), NVIDIA NIM (40 об/хв безкоштовно, 70+ моделей), Cerebras (1 млн токів/день), ключ API LongCat (50 млн токенів/день!), Cloudflare Workers AI (10 тис. нейронів/день, 50+ моделей).## Швидкий старт +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Швидкий старт ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **користувачі pnpm:**після інсталяції запустіть `pnpm approve-builds -g`, щоб увімкнути сценарії нативної збірки, необхідні для `better-sqlite3` та `@swc/core`: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > > ```bash > pnpm install -g omniroute -> pnpm approve-builds -g # Вибрати всі пакунки → схвалити +> pnpm approve-builds -g # Select all packages → approve > omniroute > ``` -Інформаційна панель відкривається за адресою `http://localhost:20128`, а базова URL-адреса API — `http://localhost:20128/v1`. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Команда | Опис | -| ----------------------- | --------------------------------------------------------------------------- | -| `omniroute` | Запустіть сервер (`PORT=20128`, API та інформаційна панель на одному порту) | -| `omniroute --порт 3000` | Установіть канонічний/API порт на 3000 | -| `omniroute --mcp` | Запустіть сервер MCP (транспорт stdio) | -| `omniroute --no-open` | Не відкривати автоматично браузер | -| `omniroute --help` | Показати довідку | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Додатковий режим розділеного порту:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -Для більшості розгортань вам потрібно лише: +For most deployments, you only need: -| Змінна | За замовчуванням | Призначення | -| ------------------------ | ----------------------------- | ---------------------------------------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600000` | Спільна базова лінія для вихідної вибірки, прихованих тайм-аутів Undici, запитів відбитків TLS і тайм-аутів запитів/проксі-серверів API | -| `STREAM_IDLE_TIMEOUT_MS` | успадковує `REQUEST_TIMEOUT_MS` | Максимальний проміжок між фрагментами потокової передачі, перш ніж OmniRoute перериває потік SSE | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -Зворотна сумісність зберігається: існуючі `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` та інші змінні часу очікування для кожного рівня все ще працюють і замінюють спільну базову лінію. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Розширені перевизначення доступні, якщо вам потрібен точніший контроль:| Змінна | За замовчуванням | Призначення | -| ---------------------------------------------- | ---------------------------------------------- | -------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | успадковує `REQUEST_TIMEOUT_MS` | Загальний тайм-аут висхідного запиту, використаний основним сигналом припинення вибірки | -| `FETCH_HEADERS_TIMEOUT_MS` | успадковує `FETCH_TIMEOUT_MS` | Обмеження часу Undici для отримання заголовків відповіді вгорі | -| `FETCH_BODY_TIMEOUT_MS` | успадковує `FETCH_TIMEOUT_MS` | Undici обмеження часу між вихідними частинами тіла (`0` вимикає його) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Тайм-аут з’єднання Undici TCP | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Час очікування неактивного сокета Undici для підтримки активності | -| `TLS_CLIENT_TIMEOUT_MS` | успадковує `FETCH_TIMEOUT_MS` | Тайм-аут для запитів відбитків TLS, зроблених через `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | успадковує `REQUEST_TIMEOUT_MS` або `30000` | Тайм-аут для пересилання проксі `/v1` з порту API на порт інформаційної панелі | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Час очікування вхідного запиту на сервері мосту API | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Час очікування вхідного заголовка на сервері мосту API | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Тайм-аут Keep-alive на сервері мосту API | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Тайм-аут бездіяльності сокета на сервері мосту API (`0` вимикає його) | +Advanced overrides are available if you need finer control: -Якщо ви запускаєте OmniRoute позаду Nginx, Caddy, Cloudflare або іншого зворотного проксі, переконайтеся, що проксі -тайм-аути також вищі, ніж ваші тайм-аути потоку/вибору OmniRoute.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Відкрийте Інформаційну панель → `Постачальники` та підключіть принаймні одного постачальника (ключ OAuth або API). -2. Відкрийте Інформаційну панель → `Кінцеві точки` та створіть ключ API. -3. (Необов’язково) Відкрийте інформаційну панель → `Комбінації` та встановіть запасний ланцюжок.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Працює з Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode та SDK, сумісними з OpenAI.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (для операцій, керованих інструментом):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -Потім підключіть свій MCP-клієнт через `stdio` та перевірте такі інструменти, як: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (для робочих процесів між агентами):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -761,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Цей набір перевіряє реальні потоки клієнтів MCP і A2A на запущену програму.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -769,14 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` -<подробиці> +
+Void Linux (`xbps-src` template) -Void Linux (шаблон `xbps-src`) - -Для користувачів Void Linux ви можете створити нативний пакет за допомогою `xbps-src`. Збережіть цей блок як `srcpkgs/omniroute/template`:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -788,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -796,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -872,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -883,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute доступний як загальнодоступний образ Docker на [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Швидкий біг:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -893,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**З файлом середовища:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Використання Docker Compose:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Підтримка інформаційної панелі для розгортання Docker тепер включає**Cloudflare Quick Tunnel**одним клацанням миші на `Інформаційна панель → Кінцеві точки`. Перший увімкне завантаження `cloudflare` лише за потреби, запускає тимчасовий тунель до вашої поточної кінцевої точки `/v1` і показує згенеровану URL-адресу `https://*.trycloudflare.com/v1` безпосередньо під вашою звичайною публічною URL-адресою. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Примітки: +Notes: -- URL-адреси швидкого тунелю є тимчасовими та змінюються після кожного перезапуску. - — Швидкі тунелі не відновлюються автоматично після перезапуску OmniRoute або контейнера. За потреби повторно ввімкніть їх на інформаційній панелі. -- Кероване встановлення наразі підтримує Linux, macOS і Windows на `x64` / `arm64`. - — У керованих швидких тунелях за замовчуванням використовується транспорт HTTP/2, щоб уникнути шумових попереджень буфера QUIC UDP у середовищах обмеженого контейнера. Установіть `CLOUDFLARED_PROTOCOL=quic` або `auto`, якщо вам потрібен інший транспорт. -- Образи Docker об’єднують корені системи ЦС і передають їх керованому `cloudflared`, що дозволяє уникнути помилок довіри TLS під час завантаження тунелю всередині контейнера. - — SQLite працює в режимі WAL. `docker stop` має завершитися, щоб OmniRoute міг перевірити останні зміни назад у `storage.sqlite`. - — У комплекті файлів Compose уже встановлено пільговий період зупинки на 40 секунд. Якщо ви запускаєте образ напряму, залиште `--stop-timeout 40` (або подібне), щоб зупинки вручну не переривали очищення завершення роботи. -- Установіть `CLOUDFLARED_BIN=/absolute/path/to/cloudflared`, якщо ви хочете, щоб OmniRoute використовував наявний двійковий файл замість його завантаження. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Використання Docker Compose з Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute можна безпечно відкрити за допомогою автоматичного забезпечення SSL Caddy. Переконайтеся, що запис DNS вашого домену вказує на IP-адресу вашого сервера.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` - -| Зображення | Тег | Розмір | Опис | +| Image | Tag | Size | Description | | ------------------------ | -------- | ------ | --------------------- | -| `diegosouzapw/omniroute` | `останній` | ~250 МБ | Останній стабільний випуск | -| `diegosouzapw/omniroute` | `1.0.3` | ~250 МБ | Поточна версія |--- +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | + +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**НОВИНКА!**OmniRoute тепер доступний як**власна настільна програма**для Windows, macOS і Linux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Запустіть OmniRoute як окрему настільну програму — для локальних моделей не потрібен ні термінал, ні браузер, ні Інтернет. Додаток на основі Electron включає: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Рідне вікно**— спеціальне вікно програми з інтеграцією в системний трей -- 🔄**Автозапуск**— запустіть OmniRoute після входу в систему -- 🔔**Власні сповіщення**— отримуйте сповіщення про вичерпання квоти або проблеми з постачальником -- ⚡**Встановлення в один клік**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Офлайн-режим**— працює повністю в автономному режимі з доданим сервером### Швидкий старт +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Швидкий старт ```bash # Development mode @@ -982,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -У згорнутому стані OmniRoute живе в панелі завдань із швидкими діями: +When minimized, OmniRoute lives in your system tray with quick actions: -- Відкрити інформаційну панель -- Змінити порт сервера -- Вийти з програми +- Open dashboard +- Change server port +- Quit application -📖 Повна документація: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Рівень | Постачальник | Вартість | Скидання квоти | Найкраще для | -| ------------------ | --------------------------- | ---------------------------------- | ----------------------------- | ------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 ПІДПИСКА** | Клод Код (Pro) | 20 доларів США на місяць | 5 годин + щотижня | Вже підписані | -| | Codex (Plus/Pro) | $20-200/міс | 5 годин + щотижня | Користувачі OpenAI | -| | Gemini CLI | **БЕЗКОШТОВНО** | 180 тис./місяць + 1 тис./день | всі! | -| | Копілот GitHub | $10-19/міс | Щомісяця | Користувачі GitHub | -| **🔑 КЛЮЧ API** | NVIDIA NIM | **БЕЗКОШТОВНО**(dev forever) | ~40 обертів за хвилину | 70+ відкритих моделей | -| | Головний мозок | **БЕЗКОШТОВНО**(1 млн ток/день) | 60K TPM / 30 RPM | Найшвидший у світі | -| | Groq | **БЕЗКОШТОВНО**(30 об/хв) | 14.4K RPD | Надшвидкий Llama/Gemma | -| | DeepSeek V3.2 | 0,27 $/1,10 $ за 1 млн | Жодного | Найкраща ціна/якість | -| | xAI Grok-4 Fast | **$0,20/$0,50 за 1 млн**🆕 | Жодного | Найшвидший + виклик інструменту, наднизький | -| | xAI Grok-4 (стандарт) | 0,20 $/1,50 $ за 1 млн. 🆕 | Жодного | Розумний флагман від xAI | -| | Містраль | Безкоштовна пробна версія + платна | Оцінка обмежена | Європейський ШІ | -| | OpenRouter | Оплата за використання | Жодного | 100+ моделей агр. | -| **💰 ДЕШЕВО** | GLM-5 (через Z.AI) 🆕 | 0,5 $/1 млн | Щодня о 10 ранку | Вихід 128K, найновіший флагман | -| | GLM-4.7 | $0,6/1 млн | Щодня о 10 ранку | Резервне копіювання бюджету | -| | MiniMax M2.5 🆕 | $0,3/1 млн вхідних даних | 5-годинний роликовий | Міркування + агентурні завдання | -| | MiniMax M2.1 | $0,2/1 млн | 5-годинний роликовий | Найдешевший варіант | -| | Kimi K2.5 (Moonshot API) 🆕 | Оплата за використання | Жодного | Прямий доступ до Moonshot API | -| | Кімі К2 | 9 $/міс квартира | 10 млн токенів/міс | Передбачувана вартість | -| **🆓 БЕЗКОШТОВНО** | Qoder | **$0** | Необмежений | 5 моделей без обмежень | -| | Квен | **$0** | Необмежений | 4 моделі без обмежень | -| | Кіро | **$0** | Необмежений | Клод Сонет/Хайку (AWS Builder) | -| | LongCat Flash-Lite 🆕 | **$0**(50 млн ток/день 🔥) | 1 RPS | Найбільша безкоштовна квота на Землі | -| | Запилення AI 🆕 | **$0**(ключ не потрібен) | 1 запит/15 с | GPT-5, Клод, DeepSeek, Лама 4 | -| | Cloudflare Workers AI 🆕 | **$0**(10 тис. нейронів/день) | ~150 повторів/день | 50+ моделей, глобальна перевага | -| | Scaleway AI 🆕 | **$0**(усього 1 млн токенів) | Оцінка обмежена | ЄС/GDPR, Qwen3 235B, Llama 70B | > 🆕**Додано нові моделі (березень 2026):**Сімейство Grok-4 Fast за $0,20/$0,50/M (тестування 1143 мс — на 30% швидше, ніж Gemini 2.5 Flash), GLM-5 через Z.AI із виведенням 128K, аргументація MiniMax M2.5, оновлена ціна DeepSeek V3.2, Kimi K2.5 через Moonshot direct API. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 Комбінований стек за $0 — повне безкоштовне налаштування:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Нульова вартість. Ніколи не припиняє кодування.**Налаштуйте це як одну комбінацію OmniRoute, і всі резервні варіанти відбуватимуться автоматично — жодного ручного перемикання.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Усі наведені нижче моделі**на 100% безкоштовні, кредитна картка не потрібна**. OmniRoute автоматично маршрутизує між ними, коли одна квота вичерпується — об’єднайте їх усі для непорушної комбінації 0 доларів США.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Модель | Префікс | Ліміт | Обмеження швидкості | -| ------------------ | ------ | ------------- | --------------------- | -| `claude-sonnet-4.5` | `kr/` |**Необмежений**| Немає повідомлень про щоденне обмеження | -| `claude-haiku-4.5` | `kr/` |**Необмежений**| Немає повідомлень про щоденне обмеження | -| `claude-opus-4.6` | `kr/` |**Необмежений**| Останній Opus через Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) -| Модель | Префікс | Ліміт | Обмеження швидкості | +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | --------------------- | +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | + +### 🟢 QODER MODELS (Free PAT via qodercli) + +| Model | Prefix | Limit | Rate Limit | | ------------------ | ------ | ------------- | --------------- | -| `kimi-k2-мислення` | `якщо/` |**Необмежений**| Немає повідомлень про обмеження | -| `qwen3-coder-plus` | `якщо/` |**Необмежений**| Немає повідомлень про обмеження | -| `deepseek-r1` | `якщо/` |**Необмежений**| Немає повідомлень про обмеження | -| `minimax-m2.1` | `якщо/` |**Необмежений**| Немає повідомлень про обмеження | -| `kimi-k2` | `якщо/` |**Необмежений**| Немає повідомлень про обмеження | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -> Рекомендований метод підключення:**Особистий маркер доступу + `qodercli`**. Браузер OAuth є -> експериментальний і вимкнений за замовчуванням, якщо не налаштовано змінні середовища `QODER_OAUTH_*`.### 🟡 QWEN MODELS (Device Code Auth) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| Модель | Префікс | Ліміт | Обмеження швидкості | -| ------------------ | ------ | ------------- | ------------------ | -| `qwen3-coder-plus` | `qw/` |**Необмежений**| Немає повідомлень про обмеження | -| `qwen3-coder-flash` | `qw/` |**Необмежений**| Немає повідомлень про обмеження | -| `qwen3-coder-next` | `qw/` |**Необмежений**| Немає повідомлень про обмеження | -| `бачення-модель` | `qw/` |**Необмежений**| Мультимодальний (зображення) |### 🟣 GEMINI CLI (Google OAuth) +### 🟡 QWEN MODELS (Device Code Auth) -| Модель | Префікс | Ліміт | Обмеження швидкості | -| ------------------------ | ------ | ---------------------------- | ------------- | -| `gemini-3-flash-preview` | `gc/` |**180K ток/місяць**+ 1K/день | Щомісячне скидання | -| `gemini-2.5-pro` | `gc/` | 180 тис./місяць (загальний пул) | Висока якість |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | ------------------- | +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Рівень | Денний ліміт | Обмеження швидкості | Примітки | -| ---------- | ------------ | ----------- | ----------------------------------------------------- | -| Безкоштовно (Dev) | Немає обмеження на маркер |**~40 об/хв**| 70+ моделей; перехід на чисті обмеження ставок у середині 2025 року | +### 🟣 GEMINI CLI (Google OAuth) -Популярні безкоштовні моделі: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | -| Рівень | Денний ліміт | Обмеження швидкості | Примітки | -| ---- | ------------------ | ---------------- | -------------------------------------------- | -| Безкоштовно |**1 млн токенів/день**| 60K TPM / 30 RPM | Найшвидший у світі LLM висновок; скидає щодня | +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) -Доступно безкоштовно: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +| Tier | Daily Limit | Rate Limit | Notes | +| ---------- | ------------ | ----------- | ------------------------------------------------------ | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -| Рівень | Денний ліміт | Обмеження швидкості | Примітки | -| ---- | ------------- | ---------------- | ---------------------------------------------- | -| Безкоштовно |**14,4 тис. RPD**| 30 обертів на хвилину на модель | Без кредитної картки; 429 на ліміті, не тарифікується | +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -Доступно безкоштовно: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -| Модель | Префікс | Щоденна безкоштовна квота | Примітки | -| ----------------------------- | ------ | ------------------ | ----------------------- | -| `LongCat-Flash-Lite` | `lc/` |**50 мільйонів токенів**💥 | Найбільша безкоштовна квота в історії | -| `LongCat-Flash-Chat` | `lc/` | 500 тисяч токенів | Багатоповоротний чат | -| `LongCat-Flash-Thinking` | `lc/` | 500 тисяч токенів | Міркування / CoT | -| `LongCat-Flash-Thinking-2601` | `lc/` | 500 тисяч токенів | Версія січня 2026 | -| `LongCat-Flash-Omni-2603` | `lc/` | 500 тисяч токенів | Мультимодальний | +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -> 100% безкоштовно під час публічної бета-версії. Зареєструйтеся на [longcat.chat](https://longcat.chat) за допомогою електронної пошти або телефону. Скидає щодня о 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -| Модель | Префікс | Обмеження швидкості | Постачальник за | +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ------------- | ---------------- | ----------------------------------------- | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | + +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` + +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 + +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | + +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. + +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| `опенай` | `pol/` | 1 запит/15 с | ГПТ-5 | -| `клод` | `pol/` | 1 запит/15 с | Антропний Клод | -| `близнюки` | `pol/` | 1 запит/15 с | Google Gemini | -| `deepseek` | `pol/` | 1 запит/15 с | DeepSeek V3 | -| `лама` | `pol/` | 1 запит/15 с | Мета Лама 4 Скаут | -| `містраль` | `pol/` | 1 запит/15 с | Містраль А. І. | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**Нульове тертя:**Без реєстрації, без ключа API. Додайте постачальника Pollinations із порожнім ключовим полем, і він запрацює негайно.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| Рівень | Щоденні нейрони | Еквівалентне використання | Примітки | -| ---- | ------------- | ----------------------------------------------- | ----------------------- | -| Безкоштовно |**10 000**| ~150 LLM resp / 500 с аудіо / 15K вбудованих | Global edge, 50+ моделей | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 -Популярні безкоштовні моделі: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (безкоштовне аудіо!), `@cf/qwen/qwen2.5-coder-15b-instruct` +| Tier | Daily Neurons | Equivalent Usage | Notes | +| ---- | ------------- | --------------------------------------- | ----------------------- | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -> Потрібен маркер API + ідентифікатор облікового запису з [dash.cloudflare.com](https://dash.cloudflare.com). Зберігайте ідентифікатор облікового запису в налаштуваннях провайдера.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -| Рівень | Безкоштовна квота | Розташування | Примітки | -| ---- | ------------- | ------------ | ---------------------------------- | -| Безкоштовно |**1 млн токенів**| 🇫🇷 Париж, ЄС | Кредитна картка не потрібна в рамках обмежень | +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -Доступно безкоштовно: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 -> Відповідає вимогам ЄС/GDPR. Отримайте ключ API на [console.scaleway.com](https://console.scaleway.com). +| Tier | Free Quota | Location | Notes | +| ---- | ------------- | ------------ | ----------------------------------- | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | ->**💡 Найкращий безкоштовний пакет (11 постачальників, 0 доларів назавжди):** +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` + +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). + +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 НЕОБМЕЖЕНО -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 млн токенів/день 🔥 -> Запилення (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — ключ не потрібен -> Qwen (qw/) → моделі qwen3-кодерів НЕОБМЕЖЕНО -> Gemini (gemini/) → Gemini 2.5 Flash — 1500 запитів/день безкоштовно -> Cloudflare AI (cf/) → 50+ моделей — 10 тис. нейронів/день -> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1 млн безкоштовних токенів (ЄС) -> Groq (groq/) → Llama/Gemma — 14,4K req/день надшвидкий -> NVIDIA NIM (nvidia/) → 70+ відкритих моделей — 40 обертів на хвилину назавжди -> Cerebras (cerebras/) → Llama/Qwen найшвидший у світі — 1 млн ток/день -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Транскрибуйте будь-яке аудіо/відео за**$0**— Deepgram лідирує з $200 безкоштовно, AssemblyAI $50 резервний варіант, Groq Whisper як необмежену екстрену резервну копію. +## 🎙️ Free Transcription Combo -| Постачальник | Безкоштовні кредити | Краща модель | Обмеження швидкості | -| ------------------ | ---------------------- | -------------------------------------------- | ---------------------------- | -| 🟢**Deepgram**|**$200 безкоштовно**(реєстрація) | `nova-3` — найкраща точність, 30+ мов | Немає ліміту RPM для безкоштовних кредитів | -| 🔵**AssemblyAI**|**$50 безкоштовно**(реєстрація) | `universal-3-pro` — розділи, почуття, PII | Немає ліміту RPM для безкоштовних кредитів | -| 🔴**Groq**|**Безкоштовно назавжди**| `whisper-large-v3` — OpenAI Whisper | 30 обертів за хвилину (швидкість обмежена) | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. -**Пропонований комбо в `/dashboard/combos`:**``` +| Provider | Free Credits | Best Model | Rate Limit | +| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | + +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Потім у `/dashboard/media` → вкладка**Транскрипція**: завантажте будь-який аудіо- чи відеофайл → виберіть комбіновану кінцеву точку → отримайте транскрипцію в підтримуваних форматах.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 створено як операційну платформу, а не просто проксі-ретранслятор.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Особливість | Що він робить | -| -------------------------------------------- | ----------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Грок-4 Швидка Сімейка** | моделі xAI за $0,20/$0,50/М — тестування 1143 мс (на 30% швидше, ніж Gemini 2.5 Flash) | -| 🧠**GLM-5 через Z.AI** | Контекст виводу 128K, $0,5/1M — найновіший флагман із сімейства GLM | -| 🔮**MiniMax M2.5** | Розуміння + агентські завдання за $0,30/1 млн — значне оновлення з M2.1 | -| 🎯**Прапор виклику інструментів для моделі** | Для кожної моделі `toolCalling: true/false` у реєстрі — AutoCombo пропускає моделі, які не підтримують інструмент | -| 🌍**Виявлення багатомовного наміру** | Ключові слова PT/ZH/ES/AR у балах AutoCombo — кращий вибір моделі для вмісту не англійською мовою | -| 📊**Резервні тести** | Справжня затримка p95 від живих запитів, комбінованих каналів — AutoCombo вивчає фактичні дані | -| 🔁**Надіслати запит на дедуплікацію** | Вікно дедуплювання на основі хешу вмісту — мультиагентний захист, запобігає повторюваним стягненням | -| 🔌**Pluggable RouterStrategy** | Розширюваний інтерфейс `RouterStrategy` — додайте спеціальну логіку маршрутизації як плагіни | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Особливість | Що він робить | -| ------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Модельний майданчик** | Сторінка інформаційної панелі для безпосереднього тестування будь-якої моделі — селектори провайдера/моделі/кінцевої точки, редактор Monaco, потокове передавання, припинення, час | -| 🔏**Зіставлення відбитків пальців CLI** | Упорядкування заголовків/телів для кожного постачальника відповідно до власних підписів CLI — перемикайте для кожного постачальника в меню «Налаштування» > «Безпека».**IP-адресу вашого проксі-сервера збережено** | -| 🤝**Підтримка ACP (протокол клієнта агента)** | Виявлення агента CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw + ще 9), створення процесу, кінцева точка `/api/acp/agents` | -| 🤖**Інформаційна панель агентів ACP** | Налагодження › Сторінка агентів — сітка з 14 агентів із статусом встановлення, версією, спеціальною формою агента для будь-якого інструменту CLI. Користувачі**OpenCode**отримують кнопку «Завантажити opencode.json», яка автоматично генерує готову до використання конфігурацію для всіх доступних моделей. | -| 🔧**Спеціальна модель маршрутизації `apiFormat`** | Спеціальні моделі з `apiFormat: "responses"` тепер правильно направляють до перекладача Responses API | -| 🏢**Ізоляція робочої області Codex** | Кілька робочих областей Codex на електронну пошту — OAuth правильно розділяє підключення за ідентифікатором робочої області | -| 🔄**Електронне автоматичне оновлення** | Додаток для комп’ютера перевіряє наявність оновлень + автоматичне встановлення після перезавантаження | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Особливість | Що він робить | -| -------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------- | -| 🔧**MCP Server (25 інструментів)** | Інструменти IDE/агента через 3 транспорти: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 основних + 3 пам'яті + 4 інструменти навичок | -| 🤝**Сервер A2A (JSON-RPC + SSE)** | Виконання завдань між агентами із синхронізацією та потоковими потоками | -| 🧭**Консолідована сторінка кінцевих точок** | Сторінка керування вкладками з вкладками Endpoint Proxy, MCP, A2A та API Endpoints | -| 🎚️**Перемикачі ввімкнення/вимкнення служби** | Перемикачі ON/OFF для MCP та A2A зі збереженням налаштувань (за замовчуванням: OFF) | -| 🛰️**MCP Runtime Heartbeat** | Справжній статус процесу (pid, безвідмовна робота, вік серцевого ритму, транспорт, режим обсягу) | -| 📋**Аудит MCP** | Фільтрувані журнали аудиту з успіхом/невдачею та ключовими атрибутами | -| 🔐**Застосування обсягу MCP** | 10 детальних дозволів для контрольованого доступу до інструментів | -| 📡**Керування життєвим циклом завдань A2A** | Список/фільтр завдань, перевірка подій/артефактів, скасування запущених завдань | -| 📋**Виявлення картки агента** | `/.well-known/agent.json` для автоматичного виявлення клієнта | -| 🧪**Тестовий джгут протоколу E2E** | Справжні MCP SDK + клієнт A2A проходять у `test:protocols:e2e` | -| ⚙️**Операційний контроль** | Комбінація перемикачів, застосування профілів стійкості, скидання вимикачів з однієї контрольної поверхні | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Особливість | Що він робить | -| ------------------------------------------------- | ---------------------------------------------------------------------------------------- | ----------------------- | -| 🎯**Розумний 4-рівневий резервний варіант** | Авто-маршрут: Підписка → Ключ API → Дешево → Безкоштовно | -| 📊**Відстеження квот у реальному часі** | Підрахунок живих токенів + скидання зворотного відліку для кожного постачальника | -| 🔄**Формат перекладу** | OpenAI ↔ Claude ↔ Gemini ↔ Відповіді з перетвореннями, безпечними для схеми | -| 👥**Підтримка кількох облікових записів** | Кілька облікових записів на постачальника з інтелектуальним вибором | -| 🔄**Автоматичне оновлення токенів** | Маркери OAuth оновлюються автоматично з повторною спробою | -| 🎨**Користувацькі комбо** | 9 стратегій балансування + резервне керування ланцюгом | -| 🌐**Wildcard Router** | `provider/*` динамічна маршрутизація | -| 🧠**Думаючи контролювати бюджет** | Наскрізні, автоматичні, користувальницькі та адаптивні межі міркування | -| 🔀**Псевдоніми моделей** | Вбудована + спеціальна модель псевдонімів і безпека міграції | -| ⚡**Погіршення фону** | Направляйте низькопріоритетні фонові завдання на дешевші моделі | -| 🧪**Розумна маршрутизація з урахуванням завдань** | Автовибір моделі за типом вмісту (кодування/бачення/аналіз/узагальнення) | -| 🔄**Робочі процеси агента A2A** | Детермінований оркестратор FSM для виконання багатокрокових агентів із збереженням стану | -| 🔀**Адаптивна маршрутизація** | Перевизначення динамічної стратегії на основі обсягу токенів і складності запиту | -| 🎲**Різноманітність постачальників** | Оцінка ентропії Шеннона, балансуючий автоматичний комбінований розподіл трафіку | -| 💬**Швидке впровадження системи** | Послідовне застосування глобальних засобів контролю поведінки | -| 📄**Сумісність API відповідей** | Повна підтримка `/v1/responses` для Codex і розширених агентських робочих процесів | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Особливість | Що він робить | -| --------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**Створення зображень** | `/v1/images/generations` з хмарними та локальними серверними частинами | -| 📐**Вбудовування** | `/v1/embeddings` для конвеєрів пошуку та RAG | -| 🎤**Транскрипція аудіо** | `/v1/audio/transcriptions` — 7 постачальників (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), автоматичне визначення мови, підтримка MP4/MP3/WAV | -| 🔊**Створення тексту в мовлення** | `/v1/audio/speech` — 10 постачальників (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) з правильними повідомленнями про помилки | -| 🎬**Створення відео** | `/v1/videos/generations` (робочі процеси ComfyUI + SD WebUI) | -| 🎵**Музичне покоління** | `/v1/music/generations` (робочі процеси ComfyUI) | -| 🛡️**Модерації** | `/v1/moderations` перевірки безпеки | -| 🔀**Переранжування** | `/v1/rerank` для підрахунку релевантності | -| 🔍**Веб-пошук**🆕 | `/v1/search` — 5 постачальників (Serper, Brave, Perplexity, Exa, Tavily), 6500+ безкоштовних/місяць, автоматичне перемикання після відмови, кеш | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Особливість | Що він робить | -| ----------------------------------------------------------- | --------------------------------------------------------------------------------------------------------- | -------------------------------- | -| 🔌**Автоматичні вимикачі** | Відключення/відновлення для кожної моделі з контролем порогових значень | -| 🎯**Моделі з урахуванням кінцевих точок** | Спеціальні моделі декларують підтримувані кінцеві точки + формат API | -| 🛡️**Anti-Thundering Herd** | Mutex + захист семафора від подій повторних спроб/швидкості | -| 🧠**Семантика + кеш підпису** | Зниження вартості/затримки за допомогою двох рівнів кешу | -| ⚡**Запит на ідемпотентність** | Дублікат вікна захисту | -| 🔒**Підробка відбитків пальців TLS** | Браузерний відбиток TLS —**зменшує виявлення ботів і позначення облікових записів** | -| 🔏**Зіставлення відбитків пальців CLI** | Відповідає власним підписам запиту CLI —**зменшує ризик заборони, зберігаючи IP-адресу проксі-сервера** | -| 🌐**IP-фільтрація** | Керування білим/чорним списком для відкритих розгортань | -| 📊**Редаговані ліміти ставок** | Конфігуровані глобальні обмеження/ліміти на рівні постачальника з постійністю | -| 📉**Витончена деградація** | Багаторівневі резервні можливості для захисту основних операцій шлюзу | -| 📜**Слід аудиту конфігурації** | Відстеження змін на основі відмінностей запобігає дрейфу операцій за допомогою простих відкатів | -| ⏳**Provider Health Sync** | Проактивний моніторинг закінчення терміну дії маркера, що запускає сповіщення перед помилками авторизації | -| 🚪**Автоматичне відключення заборонених облікових записів** | Оперативний автоматичний вимикач автоматично закриває назавжди заблоковані облікові записи токенів | -| 🔑**API Key Management + Scoping** | Безпечна видача/ротація ключів і засоби керування моделлю/постачальником | -| 👁️**Розкриття ключа API з областю дії**🆕 | Відновлення ключів API за допомогою `ALLOW_API_KEY_REVEAL` | -| 🛡️**Захищені `/models`** | Додаткова автентифікація та приховування постачальника для каталогу моделей | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Особливість | Що він робить | -| ----------------------------------------- | --------------------------------------------------------------------- | ---------------------------- | -| 📝**Запит + реєстрація проксі** | Повний журнал запитів/відповідей і проксі | -| 📉**Потокові докладні журнали**🆕 | Реконструює потоки корисного навантаження SSE в інтерфейс користувача | -| 📋**Інформаційна панель єдиних журналів** | Перегляди запитів, проксі, аудиту та консолі на одній сторінці | -| 🔍**Надіслати запит на телеметрію** | p50/p95/p99 затримка та відстеження запитів | -| 🏥**Інформаційна панель здоров’я** | Час роботи, стани поломки, блокування, статистика кешу | -| 💰**Відстеження витрат** | Контроль бюджету та видимість ціноутворення для кожної моделі | -| 📈**Аналітичні візуалізації** | Статистика використання моделі/постачальника та перегляди тенденцій | -| 🧪**Рамка оцінювання** | Тестування золотого набору з конфігурованими стратегіями матчу | -| 📡**Діагностика в реальному часі**🆕 | Обхід семантичного кешу для точного комбінованого тестування | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Особливість | Що він робить | -| ------------------------------------------- | ------------------------------------------------------------------------------------------ | --------------------- | -| 🌐**Розгортайте будь-де** | Localhost, VPS, Docker, хмарні середовища | -| 🚇**Cloudflare Tunnel**🆕 | Інтеграція Quick Tunnel одним клацанням миші з інформаційної панелі | -| 🔑**Фільтрація ключової моделі API** | Власна відповідь /v1/models, відфільтрована за допомогою призначених ролей контексту носія | -| ⚡**Розумний обхід кешу** | Настроювана евристика TTL і елементи керування примусовою перевіркою | -| 🔄**Резервне копіювання/відновлення** | Потоки експорту/імпорту та аварійного відновлення | -| 🧙**Майстер адаптації** | Кероване налаштування під час першого запуску | -| 🔧**Інформаційна панель інструментів CLI** | Налаштування в один клік для популярних інструментів кодування | -| 🎮**Модельний майданчик** | Перевірте будь-якого постачальника/модель/кінцеву точку з інформаційної панелі | -| 🔏**Перемикач відбитків пальців CLI** | Зіставлення відбитків пальців для кожного постачальника в Налаштуваннях > Безпека | -| 🌐**i18n (30 мов)** | Повна інформаційна панель + підтримка мови документів із покриттям RTL | -| 🧹**Очистити всі моделі** | Очищення списку моделей одним натисканням у деталях провайдера | -| 👁️**Елементи керування на бічній панелі**🆕 | Приховати компоненти та інтеграції в налаштуваннях зовнішнього вигляду | -| 📋**Шаблони проблем** | Стандартизовані шаблони GitHub для помилок і функцій | -| 📂**Каталог користувацьких даних** | Перевизначення `DATA_DIR` для місця зберігання | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1295,105 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -Якщо квота, швидкість або працездатність вичерпуються, OmniRoute автоматично переходить до наступного кандидата без перемикання вручну.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A можна знайти в інтерфейсі користувача та документах (не приховано) -- API статусу протоколу надають поточні робочі дані (`/api/mcp/*`, `/api/a2a/*`) -- Інформаційні панелі містять дії для операцій другого дня (комбіновані перемикачі, скидання вимикача, скасування завдання)#### Translator + validation workflow +#### Protocol management that is visible and operable -Область перекладача включає: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Playground**: вимагати перевірки трансформації -**Chat Tester**: повний запит/відповідь туди й назад -**Test Bench**: кілька випадків за один запуск -**Монітор в реальному часі**: перегляд дорожнього руху в реальному часі +#### Translator + validation workflow -Плюс перевірка протоколу з реальними клієнтами через `npm run test:protocols:e2e`. +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Довідка про інструмент, конфігурації IDE та приклади клієнтів +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A Server README](src/lib/a2a/README.md)**— Навички, методи JSON-RPC, потокове передавання та життєвий цикл завдань## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute містить вбудовану систему оцінювання для перевірки якості відповіді LLM на відповідність золотому набору. Доступ до нього через**Аналітика → Оцінки**на інформаційній панелі.### Built-in Golden Set +## 🧪 Evaluations (Evals) -Попередньо завантажений «Золотий набір OmniRoute» містить тестові випадки для: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Привітання, математика, географія, генерація коду -- Відповідність формату JSON, переклад, генерація уцінки -- Відмова безпеки (шкідливий контент), підрахунок, булева логіка### Evaluation Strategies +### Built-in Golden Set -| Стратегія | Опис | Приклад | -| ------------------ | -------------------------------------------------------------- | ------------------------------- | --- | -| `точний` | Вихідні дані повинні точно відповідати | `"4"` | -| `містить` | Вихідні дані повинні містити підрядок (незалежно від регістру) | `"Париж"` | -| `регулярний вираз` | Вихідні дані мають відповідати шаблону регулярного виразу | `"1.*2.*3"` | -| `користувацький` | Спеціальна функція JS повертає true/false | `(вивід) => output.length > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) -<подробиці> +
+🧩 MCP Setup (Model Context Protocol) -🧩 Налаштування MCP (модельний контекстний протокол) +Start MCP transport in stdio mode: -Запустіть транспорт MCP у режимі stdio:```bash +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Рекомендований процес перевірки: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Підключіть клієнт MCP через stdio. -2. Запустіть `omniroute_get_health`. -3. Запустіть `omniroute_list_combos`. -4. Відкрийте `/dashboard/mcp`, щоб підтвердити пульс, активність і аудит. - -Корисні API для автоматизації: +Useful APIs for automation: - `GET /api/mcp/status` - `GET /api/mcp/tools` - `GET /api/mcp/audit` -- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats` -<подробиці> -🤝 Налаштування A2A (Agent2Agent) + -Відкрийте агента:```bash +
+🤝 A2A Setup (Agent2Agent) + +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Надіслати завдання:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` - -Керувати життєвим циклом: +Manage lifecycle: - `GET /api/a2a/status` - `GET /api/a2a/tasks` - `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -Операційний інтерфейс користувача: +Operational UI: -- `/dashboard/a2a` для спостереження за завданням/станом/потоком і димових дій
+- `/dashboard/a2a` for task/state/stream observability and smoke actions -<подробиці> -🧪 Наскрізна перевірка протоколу + -Перевірте обидва протоколи реальними клієнтами:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Це підтверджує: +This verifies: -- Підключення клієнта MCP SDK/список/виклик -- Відкриття A2A/надсилання/потік/отримання/скасування -- Перехресна перевірка даних в аудиті MCP та API керування завданнями A2A
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs -<подробиці> + -💳 Постачальники підписки### Claude Code (Pro/Max) +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1406,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Професійна порада:**Використовуйте Opus для складних завдань, Sonnet для швидкості. OmniRoute відстежує квоту на модель!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1420,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Кожен обліковий запис Codex тепер має перемикачі політики в `Інформаційна панель -> Постачальники`: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (УВІМК./ВИМК.): примусове застосування політики порогового значення 5-годинного вікна. -- `Щотижневий` (ВКЛ./ВИМК.): застосувати політику тижневого порогового вікна. -- Порогова поведінка: коли ввімкнене вікно досягає >=90% використання, цей обліковий запис пропускається. -- Поведінка ротації: OmniRoute автоматично направляє до наступного відповідного облікового запису Codex. -- Поведінка скидання: коли проходить час `resetAt` постачальника, обліковий запис знову автоматично стає придатним. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Сценарії: +Scenarios: -- `5h ON` + `Weekly ON`: обліковий запис пропускається, коли будь-яке вікно досягає порогу. -- `5h OFF` + `Weekly ON`: лише щотижневе використання може заблокувати обліковий запис. -- `5h ON` + `Weekly OFF`: тільки 5-годинне використання може заблокувати обліковий запис. -- `resetAt` пройдено: обліковий запис автоматично повертається до ротації (без повторного ввімкнення вручну).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1445,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Найкраще:**Величезний безкоштовний рівень! Використовуйте це перед платними рівнями.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1460,74 +1662,91 @@ Models:
-<подробиці> +
+🔑 API Key Providers -🔑 Постачальники ключів API### NVIDIA NIM (FREE developer access — 70+ models) +### NVIDIA NIM (FREE developer access — 70+ models) -1. Зареєструйтеся: [build.nvidia.com](https://build.nvidia.com) -2. Отримайте безкоштовний ключ API (1000 кредитів включено) -3. Інформаційна панель → Додати постачальника → NVIDIA NIM: - - Ключ API: `nvapi-your-key` +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Моделі:**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` та ще 50+ +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -**Порада професіонала:**OpenAI-сумісний API — бездоганно працює з перекладом формату OmniRoute!### DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -1. Зареєструйтеся: [platform.deepseek.com](https://platform.deepseek.com) -2. Отримайте ключ API -3. Інформаційна панель → Додати постачальника → DeepSeek +### DeepSeek -**Моделі:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!) +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -1. Зареєструйтеся: [console.groq.com](https://console.groq.com) -2. Отримайте ключ API (включає безкоштовний рівень) -3. Інформаційна панель → Додати постачальника → Groq +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Моделі:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +### Groq (Free Tier Available!) -**Професійна порада:**Надшвидкий висновок — найкращий для кодування в реальному часі!### OpenRouter (100+ Models) +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -1. Зареєструйтеся: [openrouter.ai](https://openrouter.ai) -2. Отримайте ключ API -3. Інформаційна панель → Додати провайдера → OpenRouter +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Моделі:**Отримуйте доступ до понад 100 моделей від усіх основних постачальників за допомогою єдиного ключа API. +**Pro Tip:** Ultra-fast inference — best for real-time coding! -**Поведінка інформаційної панелі:**Моделі OpenRouter керуються з**Доступних моделей**. Ручне додавання, імпорт і автоматична синхронізація оновлюють той самий список.
+### OpenRouter (100+ Models) -<подробиці> +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -💰 Дешеві постачальники (резервні)### GLM-4.7 (Daily reset, $0.6/1M) +**Models:** Access 100+ models from all major providers through a single API key. -1. Зареєструйтеся: [Zhipu AI](https://open.bigmodel.cn/) -2. Отримайте ключ API від Coding Plan -3. Інформаційна панель → Додати ключ API: - - Постачальник: `glm` - - Ключ API: `ваш-ключ` +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -**Використовуйте:**`glm/glm-4.7` + -**Професійна порада:**План кодування пропонує 3x квоту за 1/7 вартості! Скидання щодня о 10:00.### MiniMax M2.1 (5h reset, $0.20/1M) +
+💰 Cheap Providers (Backup) -1. Зареєструйтеся: [MiniMax](https://www.minimax.io/) -2. Отримайте ключ API -3. Інформаційна панель → Додати ключ API +### GLM-4.7 (Daily reset, $0.6/1M) -**Використовуйте:**`minimax/MiniMax-M2.1` +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**Порада:**Найдешевший варіант для довгого контексту (1 млн токенів)!### Kimi K2 ($9/month flat) +**Use:** `glm/glm-4.7` -1. Підпишіться: [Moonshot AI](https://platform.moonshot.ai/) -2. Отримайте ключ API -3. Інформаційна панель → Додати ключ API +**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. -**Використовуйте:**`kimi/kimi-latest` +### MiniMax M2.1 (5h reset, $0.20/1M) -**Професійна порада:**Фіксовані 9 доларів США на місяць за 10 мільйонів токенів = 0,90 доларів США за 1 млн. ефективних витрат!
+1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key -<подробиці> +**Use:** `minimax/MiniMax-M2.1` -🆓 БЕЗКОШТОВНІ постачальники (аварійне резервне копіювання)### Qoder (5 FREE models via OAuth) +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1568,9 +1787,10 @@ Models:
-<подробиці> +
+🎨 Create Combos -🎨 Створюйте комбо### Example 1: Maximize Subscription → Cheap Backup +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1598,9 +1818,10 @@ Cost: $0 forever!
-<подробиці> +
+🔧 CLI Integration -🔧 Інтеграція CLI### Cursor IDE +### Cursor IDE ``` Settings → Models → Advanced: @@ -1611,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Використовуйте сторінку**Інструменти CLI**на інформаційній панелі для налаштування одним клацанням миші або редагуйте `~/.claude/settings.json` вручну.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1622,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Варіант 1 — Інформаційна панель (рекомендовано):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Варіант 2 — вручну:**Відредагуйте `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1639,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Примітка:**OpenClaw працює лише з локальним OmniRoute. Використовуйте `127.0.0.1` замість `localhost`, щоб уникнути проблем із вирішенням IPv6.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1653,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Крок 1:**Додайте OmniRoute як спеціального постачальника:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Крок 2:**Створіть/відредагуйте `opencode.json` у корені вашого проекту:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1679,117 +1909,130 @@ opencode } } } -```` +``` -**Крок 3:**Виберіть модель у OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Порада:**додайте будь-яку модель, доступну в кінцевій точці OmniRoute `/v1/models`, до розділу `models`. Використовуйте формат «провайдер/ідентифікатор моделі» на інформаційній панелі OmniRoute.
+ --- ## Усунення несправностей -<подробиці> -Натисніть, щоб розгорнути посібник з усунення несправностей +
+Click to expand troubleshooting guide -**"Мовна модель не надавала повідомлень"** +**"Language model did not provide messages"** -- Квота постачальника вичерпана → Перевірте систему відстеження квот на інформаційній панелі -- Рішення: скористайтеся комбінованим альтернативним варіантом або перейдіть на дешевший рівень +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**Обмеження швидкості** +**Rate limiting** -— Вичерпана квота на підписку → Повернення до GLM/MiniMax -- Додано комбо: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**Термін дії маркера OAuth минув** +**OAuth token expired** -— Автоматично оновлено OmniRoute -- Якщо проблеми не зникають: Інформаційна панель → Постачальник → Повторне підключення +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Високі витрати** +**High costs** -- Перевірте статистику використання в Інформаційній панелі → Витрати -- Переключіть основну модель на GLM/MiniMax -- Використовуйте безкоштовний рівень (Gemini CLI, Qoder) для некритичних завдань +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Порти приладової панелі/API неправильні** +**Dashboard/API ports are wrong** -- `PORT` — це канонічний базовий порт (і порт API за замовчуванням) -- `API_PORT` замінює лише сумісний із OpenAI прослухувач API -- `DASHBOARD_PORT` перевизначає лише обробник інструментальної панелі/Next.js -- Установіть `NEXT_PUBLIC_BASE_URL` для вашої інформаційної панелі/загальнодоступної URL-адреси (для зворотних викликів OAuth) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Помилки хмарної синхронізації** +**Cloud sync errors** -- Переконайтеся, що `BASE_URL` вказує на ваш запущений екземпляр -- Переконайтеся, що `CLOUD_URL` вказує на очікувану кінцеву точку хмари -- Зберігайте значення `NEXT_PUBLIC_*` узгодженими зі значеннями на стороні сервера +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Перший вхід не працює** +**First login not working** -- Перевірте `INITIAL_PASSWORD` в `.env` -- Якщо не встановлено, резервний пароль – `123456` +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Немає журналів запитів** +**No request logs** -- Артефакти запиту записуються в `DATA_DIR/call_logs/` як один файл JSON на запит -- Увімкніть захоплення конвеєра з Інформаційної панелі → Журнали → Запитувати журнали, якщо вам потрібні детальні поетапні корисні навантаження -- Встановіть `APP_LOG_TO_FILE=true`, якщо вам також потрібні журнали консолі програми в `logs/application/app.log` -- За потреби налаштуйте `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` і `CALL_LOG_MAX_ENTRIES` +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Тест з’єднання показує «Недійсне» для OpenAI-сумісних постачальників** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Багато постачальників не розкривають кінцеву точку `/models` -- OmniRoute v1.0.6+ включає резервну перевірку через завершення чату -- Переконайтеся, що базова URL-адреса містить суфікс `/v1`### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ Важливо для користувачів, які використовують OmniRoute на VPS, Docker або будь-якому віддаленому сервері**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -Постачальники**Antigravity**і**Gemini CLI**використовують**Google OAuth 2.0**. Google вимагає, щоб `redirect_uri` в потоці OAuth точно відповідав одному з попередньо зареєстрованих URI в Google Cloud Console програми. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -Облікові дані OAuth, включені в OmniRoute, зареєстровані**лише для `localhost`**. Коли ви отримуєте доступ до OmniRoute на віддаленому сервері (наприклад, `https://omniroute.myserver.com`), Google відхиляє автентифікацію за допомогою:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Вам потрібно створити**Ідентифікатор клієнта OAuth 2.0**у Google Cloud Console з URI вашого сервера.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Відкрийте Google Cloud Console** +#### Step-by-step -Перейдіть до: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Створіть новий ідентифікатор клієнта OAuth 2.0** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Натисніть**"+ Створити облікові дані"**→**"Ідентифікатор клієнта OAuth"** -- Тип програми:**"Веб-програма"** -- Назва: будь-яка, яка вам подобається (наприклад, `OmniRoute Remote`) +**2. Create a new OAuth 2.0 Client ID** -**3. Додати авторизовані URI перенаправлення** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -У полі**"Авторизовані URI перенаправлення"**додайте:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Замініть `your-server.com` доменом або IP-адресою вашого сервера (включіть порт, якщо необхідно, наприклад, `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. Збережіть і скопіюйте облікові дані** +After creating, Google will show the **Client ID** and **Client Secret**. -Після створення Google покаже**Ідентифікатор клієнта**та**Секрет клієнта**. +**5. Set environment variables** -**5. Встановити змінні середовища** +In your `.env` (or Docker environment variables): -У вашому `.env` (або змінних середовища Docker):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1798,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Перезапустіть OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Спробуйте підключитися знову** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Інформаційна панель → Постачальники → Antigravity (або Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google тепер правильно переспрямовуватиме на `https://your-server.com/callback`.--- +--- #### Temporary workaround (without custom credentials) -Якщо ви не хочете налаштовувати власні облікові дані прямо зараз, ви можете скористатися**ручним потоком URL-адреси**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute відкриває URL-адресу авторизації Google -2. Після авторизації Google намагається переспрямувати на `localhost` (це не вдається на віддаленому сервері) -3.**Скопіюйте повну URL-адресу**з адресного рядка браузера (навіть якщо сторінка не завантажується) -4. Вставте цю URL-адресу в поле, що відображається в режимі підключення OmniRoute -5. Натисніть**"Підключити"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Це працює, оскільки код авторизації в URL-адресі дійсний незалежно від того, чи завантажено сторінку переспрямування.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. -<подробиці> -🇧🇷 Versão em Português#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Провідники**Antigravity**і**Gemini CLI**використовують**Google OAuth 2.0**для автентифікації. Google вимагає, щоб `redirect_uri` не використовував fluxo OAuth, щоб**exatamente**мати URI перед кадастрадами без додатка Google Cloud Console. +
+🇧🇷 Versão em Português -Як повноваження OAuth embutidas, OmniRoute не встановлено в кадастрадах**apenas para `localhost`**. Якщо ви маєте доступ до OmniRoute у віддаленому сервері (наприклад: `https://omniroute.meuservidor.com`), або Google rejeita a autenticação com:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Потрібно точно написати**Ідентифікатор клієнта OAuth 2.0**у Google Cloud Console через URI вашого сервера.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Доступ до Google Cloud Console** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -**2. Crie um novo OAuth 2.0 ID клієнта** +**2. Crie um novo OAuth 2.0 Client ID** -- Натисніть**"+ Створити облікові дані"**→**"Ідентифікатор клієнта OAuth"** -- Tipo de aplicativo:**"Веб-програма"** -- Назва: escolha qualquer nome (наприклад: `OmniRoute Remote`) +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Додайте як авторизовані URI перенаправлення** +**3. Adicione as Authorized Redirect URIs** -Без поля**"Авторизовані URI перенаправлення"**, додайте:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Замініть `seu-servidor.com` домінування або IP на свій сервер (включно з необхідним портом, наприклад: `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. Зберегти електронну копію як ідентифікацію** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Наприклад, Google показує**Ідентифікатор клієнта**і**Секрет клієнта**. +**5. Configure as variáveis de ambiente** -**5. Налаштувати як variáveis de ambiente** +No seu `.env` (ou nas variáveis de ambiente do Docker): -Немає `.env` (або нас варіативних умов Docker):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1877,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reinicie o OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute - -```` +``` **7. Tente conectar novamente** -Інформаційна панель → Постачальники → Антигравітація (або Gemini CLI) → OAuth +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Agora o Google redirectionará corretamente para `https://seu-servidor.com/callback` and a authenticação funcionará.--- +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. + +--- #### Workaround temporário (sem configurar credenciais próprias) -Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo**manual de URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. OmniRoute скидає URL-адресу авторизації Google -2. Якщо ви авторизуєтеся, або перенаправте Google для локального хосту (якщо сервер не віддалений) -3.**Скопіюйте повну URL-адресу**для переходу до вашого браузера (повідомте, що сторінка не створена) +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) 4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute -5. Натисніть**"Підключити"** +5. Clique em **"Connect"** -> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1915,64 +2171,73 @@ Se não quiser criar credenciais próprias agora, ainda é possível usar o flux ## 🛠️ Tech Stack -<подробиці> -Натисніть, щоб розгорнути деталі стеку технологій +
+Click to expand tech stack details --**Серед виконання**: Node.js 18–22 LTS (⚠️ Node.js 24+**не підтримується**— власні двійкові файли `better-sqlite3` несумісні) --**Мова**: TypeScript 5.9 —**100% TypeScript**для `src/` і `open-sse/` (нуль `any` в основних модулях, починаючи з версії 2.0) --**Framework**: Next.js 16 + React 19 + Tailwind CSS 4 --**База даних**: LowDB (JSON) + SQLite (стан домену + журнали проксі + аудит MCP + рішення про маршрутизацію) --**Схеми**: Zod (перевірка введення/виведення інструменту MCP, контракти API) --**Протоколи**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Потокове передавання**: події, надіслані сервером (SSE) --**Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization --**Тестування**: програма для виконання тестів Node.js + Vitest (900+ тестів, включаючи блоки, інтеграцію, E2E) --**CI/CD**: дії GitHub (автоматична публікація npm + Docker Hub після випуску) --**Веб-сайт**: [omniroute.online](https://omniroute.online) --**Пакет**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Стійкість**: автоматичний вимикач, експоненціальна віддача, захист від гриму, підробка TLS, автоматичне самовідновлення комбо
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Документація -| Документ | Опис | -| ---------------------------------------------- | -------------------------------------------------- | -| [Посібник користувача](docs/USER_GUIDE.md) | Постачальники, комбо, інтеграція CLI, розгортання | -| [Довідка по API](docs/API_REFERENCE.md) | Усі кінцеві точки з прикладами | -| [MCP-сервер](open-sse/mcp-server/README.md) | 16 інструментів MCP, конфігурації IDE, клієнти Python/TS/Go | -| [Сервер A2A](src/lib/a2a/README.md) | Протокол JSON-RPC 2.0, навички, потокове передавання, керування завданнями | -| [Auto-Combo Engine](docs/auto-combo.md) | 6-факторна оцінка, пакети режимів, самовідновлення | -| [Усунення несправностей](docs/TROUBLESHOOTING.md) | Загальні проблеми та рішення | -| [Архітектура](docs/ARCHITECTURE.md) | Архітектура системи та внутрішні | -| [Внесок](CONTRIBUTING.md) | Розробка установки та рекомендацій | -| [Специфікація OpenAPI](docs/openapi.yaml) | Специфікація OpenAPI 3.0 | -| [Політика безпеки](SECURITY.md) | Повідомлення про вразливості та методи безпеки | -| [Розгортання віртуальної машини](docs/VM_DEPLOYMENT_GUIDE.md) | Повний посібник: налаштування VM + nginx + Cloudflare | -| [Галерея функцій](docs/FEATURES.md) | Огляд інформаційної панелі зі знімками екрана | -| [Контрольний список випуску](docs/RELEASE_CHECKLIST.md) | Етапи перевірки перед випуском |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute має**заплановано понад 210 функцій**на кількох етапах розробки. Ось ключові області: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Категорія | Заплановані особливості | Основні моменти | -| ----------------------------- | ---------------- | ------------------------------------------------------------------------------------- | -| 🧠**Маршрутизація та інтелект**| 25+ | Маршрутизація з найменшою затримкою, маршрутизація на основі тегів, попередній перегляд квот, вибір облікового запису P2C | -| 🔒**Безпека та відповідність**| 20+ | Захист SSRF, маскування облікових даних, обмеження швидкості для кінцевої точки, визначення обсягу ключа керування | -| 📊**Спостережливість**| 15+ | Інтеграція OpenTelemetry, моніторинг квот у реальному часі, відстеження витрат на модель | -| 🔄**Інтеграція постачальників**| 20+ | Реєстр динамічної моделі, час відновлення провайдера, Codex із кількома обліковими записами, розбір квоти Copilot | -| ⚡**Виконання**| 15+ | Подвійний рівень кешу, кеш запитів, кеш відповідей, потокове підтримання активності, пакетний API | -| 🌐**Екосистема**| 10+ | API WebSocket, гаряче перезавантаження конфігурації, розподілене сховище конфігурацій, комерційний режим |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**Інтеграція OpenCode**— власна підтримка постачальника для IDE кодування OpenCode AI -- 🔗**TRAE Integration**— повна підтримка інфраструктури розробки TRAE AI -- 📦**Batch API**— асинхронна пакетна обробка масових запитів -- 🎯**Маршрутизація на основі тегів**— Маршрутизація запитів на основі спеціальних тегів і метаданих -- 💰**Стратегія найнижчої вартості**— автоматично вибирайте найдешевшого доступного постачальника +### 🔜 Coming Soon -> 📝 Повні специфікації функцій доступні в [`docs/new-features/`](docs/new-features/) (217 детальних специфікацій)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1980,18 +2245,20 @@ OmniRoute має**заплановано понад 210 функцій**на к ### How to Contribute -1. Розгалужте репозиторій -2. Створіть свою гілку функцій (`git checkout -b feature/amazing-feature`) -3. Зафіксуйте свої зміни (`git commit -m 'Додати чудову функцію'`) -4. Надішліть до гілки (`git push origin feature/amazing-feature`) -5. Відкрийте Pull Request +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Дивіться [CONTRIBUTING.md](CONTRIBUTING.md), щоб отримати докладні вказівки.### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -2003,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Особлива подяка**[9router](https://github.com/decolua/9router)**від**[decolua](https://github.com/decolua)**— оригінальному проекту, який надихнув цей форк. OmniRoute спирається на цю неймовірну основу завдяки додатковим функціям, мультимодальним API і повному перепису TypeScript. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Особлива подяка**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— оригінальній реалізації Go, яка надихнула цей порт JavaScript.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Ліцензія -Ліцензія Массачусетського технологічного інституту – див. [ЛІЦЕНЗІЯ](ЛІЦЕНЗІЯ) для отримання додаткової інформації.--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/uk-UA/docs/ARCHITECTURE.md b/docs/i18n/uk-UA/docs/ARCHITECTURE.md index dd2e66b974..b82cb25c11 100644 --- a/docs/i18n/uk-UA/docs/ARCHITECTURE.md +++ b/docs/i18n/uk-UA/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Останнє оновлення: 2026-03-28_## Executive Summary -OmniRoute — це локальний шлюз штучного інтелекту та інформаційна панель, побудована на Next.js. -Він надає єдину кінцеву точку, сумісну з OpenAI (`/v1/*`), і направляє трафік між декількома вихідними постачальниками з перекладом, резервним варіантом, оновленням маркерів і відстеженням використання. -Основні можливості: +_Last updated: 2026-03-28_ -- OpenAI-сумісна поверхня API для CLI/інструментів (28 постачальників) -- Переклад запитів/відповідей між форматами постачальників -- Запасна комбінована модель (багатомодельна послідовність) -- Запасний варіант на рівні облікового запису (декілька облікових записів на постачальника) -- OAuth + API-ключ управління підключенням провайдера -- Генерація вбудовування через `/v1/embeddings` (6 постачальників, 9 моделей) -- Створення зображень через `/v1/images/generations` (4 постачальники, 9 моделей) - — Розбір тегів мислення (`...`) для моделей міркування - — Дезінфекція відповіді для суворої сумісності з OpenAI SDK - — Нормалізація ролі (розробник→система, система→користувач) для сумісності між постачальниками -- Перетворення структурованого виводу (json_schema → Gemini responseSchema) -- Локальна постійність для провайдерів, ключів, псевдонімів, комбо, налаштувань, ціноутворення -- Відстеження використання/вартості та реєстрація запитів -- Додаткова хмарна синхронізація для синхронізації кількох пристроїв/станів - — Список дозволених/чорних IP-адрес для контролю доступу до API -- Продумане управління бюджетом (прохідний/автоматичний/спеціальний/адаптивний) -- Оперативна ін'єкція глобальної системи -- Відстеження сесії та відбитки пальців -- Розширене обмеження швидкості для кожного облікового запису за допомогою профілів постачальника -- Схема автоматичного вимикача для стійкості провайдера -- Захист стада від грому з блокуванням м'ютексу - — Кеш дедуплікації запитів на основі підпису -- Рівень домену: доступність моделі, правила вартості, резервна політика, політика блокування -- Постійність стану домену (скрізний кеш SQLite для резервних копій, бюджетів, блокувань, автоматичних вимикачів) -- Механізм політики для централізованої оцінки запитів (блокування → бюджет → резервний варіант) -- Запит телеметрії з агрегацією затримок p50/p95/p99 -- Ідентифікатор кореляції (X-Request-Id) для наскрізного відстеження -- Журнал аудиту відповідності з відмовою для кожного ключа API -- Eval framework для забезпечення якості LLM - — Панель інструментів інтерфейсу Resilience зі статусом автоматичного вимикача в режимі реального часу -- Модульні постачальники OAuth (12 окремих модулів у `src/lib/oauth/providers/`) +## Executive Summary -Основна модель середовища виконання: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- Маршрути додатків Next.js у `src/app/api/*` реалізують як API панелі керування, так і API сумісності -- Спільне ядро SSE/маршрутизації в `src/sse/*` + `open-sse/*` обробляє виконання провайдера, переклад, потокове передавання, відкат і використання## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Час виконання локального шлюзу -- API керування інформаційною панеллю -- Автентифікація постачальника та оновлення маркера -- Запит на переклад і потокове передавання SSE - — Локальний стан + постійність використання - — Додаткова синхронізація з хмарою### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Реалізація хмарної служби за `NEXT_PUBLIC_CLOUD_URL` -- Площина SLA/контроль постачальника поза локальним процесом -- Самі зовнішні двійкові файли CLI (Claude CLI, Codex CLI тощо)## Dashboard Surface (Current) +### Out of Scope -Головні сторінки в `src/app/(dashboard)/dashboard/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — швидкий старт + огляд провайдера -- `/dashboard/endpoint` — проксі кінцевої точки + MCP + A2A + вкладки кінцевої точки API -- `/dashboard/providers` — підключення та облікові дані провайдера -- `/dashboard/combos` — комбіновані стратегії, шаблони, правила маршрутизації моделей -- `/dashboard/costs` — агрегація витрат і видимість цін -- `/dashboard/analytics` — аналітика та оцінка використання -- `/dashboard/limits` — елементи керування квотою/швидкістю -- `/dashboard/cli-tools` — адаптація CLI, визначення часу виконання, створення конфігурації -- `/dashboard/agents` — виявлені агенти ACP + реєстрація спеціального агента -- `/dashboard/media` — майданчик для зображень/відео/музики -- `/dashboard/search-tools` — тестування пошукового провайдера та історія -- `/dashboard/health` — час роботи, автоматичні вимикачі, обмеження швидкості -- `/dashboard/logs` — журнали запитів/проксі/аудиту/консолі -- `/dashboard/settings` — вкладки системних налаштувань (загальні, маршрутизація, комбіновані параметри за замовчуванням тощо) -- `/dashboard/api-manager` — життєвий цикл ключа API та дозволи моделі## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Основні каталоги: +Main directories: -- `src/app/api/v1/*` і `src/app/api/v1beta/*` для API сумісності -- `src/app/api/*` для API керування/конфігурації -- Далі перезаписує в `next.config.mjs` карту `/v1/*` на `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Важливі маршрути сумісності: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` — включає власні моделі з `custom: true` -- `src/app/api/v1/embeddings/route.ts` — генерація вбудовування (6 провайдерів) -- `src/app/api/v1/images/generations/route.ts` — генерація зображень (4+ постачальники, включно з Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — спеціальний чат для кожного постачальника -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — виділені вбудовування для кожного постачальника -- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — виділені зображення для кожного постачальника +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` -- `src/app/api/v1beta/models/[...шлях]/route.ts` +- `src/app/api/v1beta/models/[...path]/route.ts` -Домени керування: +Management domains: - Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` -- Постачальники/підключення: `src/app/api/providers*` -- Вузли постачальника: `src/app/api/provider-nodes*` -- Спеціальні моделі: `src/app/api/provider-models` (GET/POST/DELETE) -- Каталог моделей: `src/app/api/models/route.ts` (GET) -- Конфігурація проксі: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- Ключі/псевдоніми/комбінації/ціни: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- Використання: `src/app/api/usage/*` -- Синхронізація/хмара: `src/app/api/sync/*`, `src/app/api/cloud/*` -- Допоміжні інструменти CLI: `src/app/api/cli-tools/*` -- IP-фільтр: `src/app/api/settings/ip-filter` (GET/PUT) -- Бюджет мислення: `src/app/api/settings/thinking-budget` (GET/PUT) -- Системний запит: `src/app/api/settings/system-prompt` (GET/PUT) -- Сеанси: `src/app/api/sessions` (GET) -- Обмеження швидкості: `src/app/api/rate-limits` (GET) -- Стійкість: `src/app/api/resilience` (GET/PATCH) — профілі постачальників, автоматичний вимикач, граничний стан швидкості -- Скидання стійкості: `src/app/api/resilience/reset` (POST) — скидання вимикачів + час відновлення -- Статистика кешу: `src/app/api/cache/stats` (GET/DELETE) -- Доступність моделі: `src/app/api/models/availability` (GET/POST) -- Телеметрія: `src/app/api/telemetry/summary` (GET) -- Бюджет: `src/app/api/usage/budget` (GET/POST) -- Резервні ланцюжки: `src/app/api/fallback/chains` (GET/POST/DELETE) -- Аудит відповідності: `src/app/api/compliance/audit-log` (GET) -- Оцінки: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- Політики: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -Основні модулі потоку: +## 2) SSE + Translation Core -- Запис: `src/sse/handlers/chat.ts` -- Оркестровка ядра: `open-sse/handlers/chatCore.ts` - — Адаптери виконання постачальника: `open-sse/executors/*` -- Виявлення формату/конфігурація постачальника: `open-sse/services/provider.ts` -- Розбір/вирішення моделі: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Логіка резервного облікового запису: `open-sse/services/accountFallback.ts` - — Реєстр перекладів: `open-sse/translator/index.ts` -- Перетворення потоку: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- Вилучення/нормалізація використання: `open-sse/utils/usageTracking.ts` - — Парсер тегів Think: `open-sse/utils/thinkTagParser.ts` -- Обробник вбудовування: `open-sse/handlers/embeddings.ts` -- Реєстр постачальника вбудовування: `open-sse/config/embeddingRegistry.ts` - — Обробник створення зображення: `open-sse/handlers/imageGeneration.ts` -- Реєстр постачальників зображень: `open-sse/config/imageRegistry.ts` -- Очищення відповіді: `open-sse/handlers/responseSanitizer.ts` - — Нормалізація ролі: `open-sse/services/roleNormalizer.ts` +Main flow modules: -Послуги (бізнес-логіка): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- Вибір/оцінка облікового запису: `open-sse/services/accountSelector.ts` -- Управління життєвим циклом контексту: `open-sse/services/contextManager.ts` - — Примусовий IP-фільтр: `open-sse/services/ipFilter.ts` -- Відстеження сесії: `open-sse/services/sessionManager.ts` -- Дедуплікація запиту: `open-sse/services/signatureCache.ts` -- Ін'єкція системної підказки: `open-sse/services/systemPrompt.ts` -- Розумне управління бюджетом: `open-sse/services/thinkingBudget.ts` -- Маршрутизація моделі підстановок: `open-sse/services/wildcardRouter.ts` - — Керування обмеженнями тарифів: `open-sse/services/rateLimitManager.ts` -- Автоматичний вимикач: `open-sse/services/circuitBreaker.ts` +Services (business logic): -Модулі доменного рівня: +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- Доступність моделі: `src/lib/domain/modelAvailability.ts` -- Правила/бюджети витрат: `src/lib/domain/costRules.ts` -- Резервна політика: `src/lib/domain/fallbackPolicy.ts` -- Комбінований розпізнавач: `src/lib/domain/comboResolver.ts` -- Політика блокування: `src/lib/domain/lockoutPolicy.ts` -- Механізм політики: `src/domain/policyEngine.ts` — централізоване блокування → бюджет → резервна оцінка -- Каталог кодів помилок: `src/lib/domain/errorCodes.ts` -- Ідентифікатор запиту: `src/lib/domain/requestId.ts` -- Час очікування отримання: `src/lib/domain/fetchTimeout.ts` -- Запит телеметрії: `src/lib/domain/requestTelemetry.ts` -- Відповідність/аудит: `src/lib/domain/compliance/index.ts` -- Бігун Eval: `src/lib/domain/evalRunner.ts` -- Постійність стану домену: `src/lib/db/domainState.ts` — SQLite CRUD для резервних ланцюжків, бюджетів, історії витрат, стану блокування, автоматичних вимикачів +Domain layer modules: -Модулі постачальника OAuth (12 окремих файлів у `src/lib/oauth/providers/`): +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -- Індекс реєстру: `src/lib/oauth/providers/index.ts` -- Окремі постачальники: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` -- Тонка оболонка: `src/lib/oauth/providers.ts` — реекспорт з окремих модулів## 3) Persistence Layer +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -База даних первинного стану (SQLite): +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -- Базовий інфра: `src/lib/db/core.ts` (better-sqlite3, міграції, WAL) - — Реекспорт фасаду: `src/lib/localDb.ts` (тонкий рівень сумісності для абонентів) -- файл: `${DATA_DIR}/storage.sqlite` (або `$XDG_CONFIG_HOME/omniroute/storage.sqlite`, якщо встановлено, інакше `~/.omniroute/storage.sqlite`) -- сутності (таблиці + простори імен KV): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +## 3) Persistence Layer -Стійкість використання: +Primary state DB (SQLite): -- фасад: `src/lib/usageDb.ts` (розкладені модулі в `src/lib/usage/*`) -- Таблиці SQLite в `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` -- необов'язкові артефакти файлів залишаються для сумісності/налагодження (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- застарілі файли JSON переносяться до SQLite за допомогою міграції під час запуску, якщо вони присутні +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -БД стану домену (SQLite): +Usage persistence: -- `src/lib/db/domainState.ts` — операції CRUD для стану домену -- Таблиці (створені в `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` -- Шаблон кешу наскрізного запису: Карти в пам'яті є авторитетними під час виконання; мутації записуються синхронно в SQLite; стан відновлюється з БД при холодному запуску## 4) Auth + Security Surfaces +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- Автентифікація файлів cookie інформаційної панелі: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- Генерація/перевірка ключа API: `src/shared/utils/apiKey.ts` -- Секрети постачальника зберігалися в записах `providerConnections` -- Підтримка вихідного проксі-сервера через `open-sse/utils/proxyFetch.ts` (env vars) і `open-sse/utils/networkProxy.ts` (налаштовується для кожного постачальника або глобально)## 5) Cloud Sync +Domain State DB (SQLite): -- Ініціалізація планувальника: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Періодичне завдання: `src/shared/services/cloudSyncScheduler.ts` -- Періодичне завдання: `src/shared/services/modelSyncScheduler.ts` -- Контрольний маршрут: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Запасні рішення приймаються open-sse/services/accountFallback.ts за допомогою кодів стану та евристики повідомлень про помилки. Комбінована маршрутизація додає один додатковий захист: 400-і з областю постачальника, такі як помилки блокування вмісту вгорі та перевірки ролі, розглядаються як локальні помилки моделі, тому пізніші комбіновані цілі все ще можуть працювати.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -Оновлення під час живого трафіку виконується всередині `open-sse/handlers/chatCore.ts` за допомогою виконавця `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -Періодичну синхронізацію запускає `CloudSyncScheduler`, коли хмару ввімкнено.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -Файли фізичного зберігання: +Physical storage files: -- основна база даних середовища виконання: `${DATA_DIR}/storage.sqlite` -- рядки журналу запитів: `${DATA_DIR}/log.txt` (компатія/налагодження артефакту) -- структуровані архіви корисного навантаження викликів: `${DATA_DIR}/call_logs/` -- додатковий перекладач/сеанси налагодження запитів: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,206 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: API сумісності -- `src/app/api/v1/providers/[provider]/*`: виділені маршрути для кожного постачальника (чат, вбудовування, зображення) -- `src/app/api/providers*`: CRUD провайдера, перевірка, тестування -- `src/app/api/provider-nodes*`: настроюване керування сумісним вузлом -- `src/app/api/provider-models`: керування власною моделлю (CRUD) -- `src/app/api/models/route.ts`: API каталогу моделей (псевдоніми + спеціальні моделі) -- `src/app/api/oauth/*`: потоки OAuth/код пристрою -- `src/app/api/keys*`: життєвий цикл локального ключа API -- `src/app/api/models/alias`: керування псевдонімами -- `src/app/api/combos*`: резервне керування комбо -- `src/app/api/pricing`: заміна ціноутворення для розрахунку вартості -- `src/app/api/settings/proxy`: налаштування проксі (GET/PUT/DELETE) -- `src/app/api/settings/proxy/test`: перевірка підключення вихідного проксі (POST) -- `src/app/api/usage/*`: API використання та журналів -- `src/app/api/sync/*` + `src/app/api/cloud/*`: хмарна синхронізація та помічники, спрямовані на хмару -- `src/app/api/cli-tools/*`: локальні автори/перевірки налаштувань CLI -- `src/app/api/settings/ip-filter`: список дозволених/чорних IP-адрес (GET/PUT) -- `src/app/api/settings/thinking-budget`: конфігурація бюджету маркера мислення (GET/PUT) -- `src/app/api/settings/system-prompt`: глобальне системне підказка (GET/PUT) -- `src/app/api/sessions`: список активних сеансів (GET) -- `src/app/api/rate-limits`: стан обмеження ставки для кожного облікового запису (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: розбір запитів, комбо-обробка, цикл вибору облікового запису -- `open-sse/handlers/chatCore.ts`: переклад, розсилка виконавця, обробка повторів/оновлень, налаштування потоку -- `open-sse/executors/*`: залежна від провайдера мережа та поведінка формату### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: реєстр перекладачів та оркестровка -- Перекладач запитів: `open-sse/translator/request/*` -- Транслятори відповідей: `open-sse/translator/response/*` -- Константи формату: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: постійна конфігурація/стан і збереження домену на SQLite -- `src/lib/localDb.ts`: реекспорт сумісності для модулів БД -- `src/lib/usageDb.ts`: фасад історії використання/журналів викликів поверх таблиць SQLite## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Кожен провайдер має спеціалізований виконавець, що розширює `BaseExecutor` (у `open-sse/executors/base.ts`), який забезпечує побудову URL-адреси, побудову заголовка, повторну спробу з експоненційною відстрочкою, перехоплювачі оновлення облікових даних і метод оркестровки `execute()`. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Виконавець | Постачальник(и) | Спеціальна обробка | -| ----------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------------------------------------------- | -| `Виконавець за замовчуванням` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Конфігурація динамічної URL-адреси/заголовка для кожного постачальника | -| `AntgravityExecutor` | Антигравітація Google | Ідентифікатори користувацьких проектів/сеансів, повторна спроба після аналізу | -| `CodexExecutor` | OpenAI Codex | Впроваджує системні інструкції, змушує міркувати | -| `CursorExecutor` | Курсор IDE | Протокол ConnectRPC, кодування Protobuf, підпис запиту через контрольну суму | -| `GithubExecutor` | Копілот GitHub | Оновлення маркера Copilot, заголовки, що імітують VSCode | -| `KiroExecutor` | AWS CodeWhisperer/Kiro | Двійковий формат AWS EventStream → Перетворення SSE | -| `GeminiCLIExecutor` | Gemini CLI | Цикл оновлення маркера Google OAuth | +### Persistence -Усі інші постачальники (включно з настроюваними сумісними вузлами) використовують `DefaultExecutor`.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Постачальник | Формат | Авторизація | Потік | Непотоковий | Токен Оновити | Використання API | -| ---------------- | ---------------- | ---------------------- | ---------------- | ----------- | ------------- | ------------------------- | ------------------------------ | -| Клод | Клод | Ключ API / OAuth | ✅ | ✅ | ✅ | ⚠️ Лише адміністратор | -| Близнюки | близнюки | Ключ API / OAuth | ✅ | ✅ | ✅ | ⚠️ Хмарна консоль | -| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Хмарна консоль | -| Антигравітація | антигравітація | OAuth | ✅ | ✅ | ✅ | ✅ Повна квота API | -| OpenAI | openai | Ключ API | ✅ | ✅ | ❌ | ❌ | -| Кодекс | openai-відповіді | OAuth | ✅ примусовий | ❌ | ✅ | ✅ Обмеження тарифів | -| Копілот GitHub | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Знімки квот | -| Курсор | курсор | Власна контрольна сума | ✅ | ✅ | ❌ | ❌ | -| Кіро | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Обмеження використання | -| Квен | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ За запитом | -| Qoder | openai | OAuth (базовий) | ✅ | ✅ | ✅ | ⚠️ За запитом | -| OpenRouter | openai | Ключ API | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | Клод | Ключ API | ✅ | ✅ | ❌ | ❌ | -| DeepSeek | openai | Ключ API | ✅ | ✅ | ❌ | ❌ | -| Groq | openai | Ключ API | ✅ | ✅ | ❌ | ❌ | -| xAI (Грок) | openai | Ключ API | ✅ | ✅ | ❌ | ❌ | -| Містраль | openai | Ключ API | ✅ | ✅ | ❌ | ❌ | -| Розгубленість | openai | Ключ API | ✅ | ✅ | ❌ | ❌ | -| Разом AI | openai | Ключ API | ✅ | ✅ | ❌ | ❌ | -| Феєрверк AI | openai | Ключ API | ✅ | ✅ | ❌ | ❌ | -| Головний мозок | openai | Ключ API | ✅ | ✅ | ❌ | ❌ | -| Cohere | openai | Ключ API | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | openai | Ключ API | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Виявлені вихідні формати включають: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. -- `опенай` -- `openai-відповіді` -- "клод". -- "близнюки". +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | -Цільові формати включають: +All other providers (including custom compatible nodes) use the `DefaultExecutor`. -- Чат/Відповіді OpenAI -- Клод -- Gemini/Gemini-CLI/Антигравітаційна оболонка -- Кіро -- Курсор +## Provider Compatibility Matrix -Для перекладу використовується**OpenAI як центральний формат**— усі перетворення проходять через OpenAI як проміжний:``` +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: + +- `openai` +- `openai-responses` +- `claude` +- `gemini` + +Target formats include: + +- OpenAI chat/Responses +- Claude +- Gemini/Gemini-CLI/Antigravity envelope +- Kiro +- Cursor + +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Переклади вибираються динамічно на основі форми вихідного корисного навантаження та цільового формату постачальника. +Additional processing layers in the translation pipeline: -Додаткові рівні обробки в конвеєрі перекладу: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Дезінфікація відповідей**— видаляє нестандартні поля з відповідей у форматі OpenAI (як потокових, так і не потокових), щоб забезпечити сувору відповідність SDK --**Нормалізація ролі**— перетворює `розробник` в `систему` для цілей, що не належать до OpenAI; об’єднує `system` → `user` для моделей, які відхиляють системну роль (GLM, ERNIE) --**Вилучення тегів мислення**— аналізує блоки `...` із вмісту в поле `reasoning_content` --**Структурований вихід**— перетворює OpenAI `response_format.json_schema` на `responseMimeType` + `responseSchema` Gemini## Supported API Endpoints +## Supported API Endpoints -| Кінцева точка | Формат | Обробник | -| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------ | -| `POST /v1/chat/completions` | Чат OpenAI | `src/sse/handlers/chat.ts` | -| `POST /v1/messages` | Повідомлення Клода | Той самий обробник (визначено автоматично) | -| `POST /v1/responses` | Відповіді OpenAI | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/embeddings` | Вбудовування OpenAI | `open-sse/handlers/embeddings.ts` | -| `GET /v1/embeddings` | Список моделей | Маршрут API | -| `POST /v1/images/generations` | Зображення OpenAI | `open-sse/handlers/imageGeneration.ts` | -| `GET /v1/images/generations` | Список моделей | Маршрут API | -| `POST /v1/providers/{provider}/chat/completions` | Чат OpenAI | Виділено для кожного постачальника з перевіркою моделі | -| `POST /v1/providers/{provider}/embeddings` | Вбудовування OpenAI | Виділено для кожного постачальника з перевіркою моделі | -| `POST /v1/providers/{provider}/images/generations` | Зображення OpenAI | Виділено для кожного постачальника з перевіркою моделі | -| `POST /v1/messages/count_tokens` | Клод Токен Підрахунок | Маршрут API | -| `GET /v1/models` | Список моделей OpenAI | Маршрут API (чат + вбудовування + зображення + спеціальні моделі) | -| `GET /api/models/catalog` | Каталог | Усі моделі згруповані за постачальником + тип | -| `POST /v1beta/models/*:streamGenerateContent` | Близнюки рідні | Маршрут API | -| `GET/PUT/DELETE /api/settings/proxy` | Конфігурація проксі | Налаштування мережевого проксі | -| `POST /api/settings/proxy/test` | З'єднання проксі | Кінцева точка перевірки справності/з’єднання проксі | -| `GET/POST/DELETE /api/provider-models` | Моделі постачальників | Метадані моделі постачальника, що підтримують спеціальні та керовані доступні моделі |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -Обхідний обробник (`open-sse/utils/bypassHandler.ts`) перехоплює відомі "викидні" запити від Claude CLI — пінг розігріву, вилучення заголовків і підрахунок токенів — і повертає**фальшиву відповідь**, не споживаючи токени постачальника вгору. Це спрацьовує лише тоді, коли `User-Agent` містить `claude-cli`.## Request Logger Pipeline +## Bypass Handler -Реєстратор запитів (`open-sse/utils/requestLogger.ts`) забезпечує 7-етапний конвеєр журналювання налагодження, вимкнений за замовчуванням, увімкнений за допомогою `ENABLE_REQUEST_LOGS=true`:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Файли записуються в /logs//` для кожного сеансу запиту.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- час відновлення облікового запису постачальника через тимчасові помилки/помилки швидкості/автентифікації -- резервний обліковий запис перед невдалим запитом -- резервна комбінована модель, коли поточний шлях моделі/постачальника вичерпано## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- попередня перевірка та оновлення з повторною спробою для оновлюваних постачальників -- Повторна спроба 401/403 після спроби оновлення в основному шляху## 3) Stream Safety +## 2) Token Expiry -- контролер потоку з відключенням -- потік перекладу зі змивом наприкінці потоку та обробкою `[DONE]` -- резервна оцінка використання, якщо метадані використання постачальника відсутні## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- виникають помилки синхронізації, але локальне виконання продовжується -- планувальник має логіку повторної спроби, але періодичне виконання наразі викликає синхронізацію з одноразовою спробою за замовчуванням## 5) Data Integrity +## 3) Stream Safety -— Міграції схем SQLite та автооновлення під час запуску +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -- застарілий JSON → шлях сумісності міграції SQLite## Observability and Operational Signals +## 4) Cloud Sync Degradation -Джерела видимості під час виконання: +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -- журнали консолі з `src/sse/utils/logger.ts` -- агрегати використання за запитом у SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- чотириетапний детальний запис корисного навантаження в SQLite (`request_detail_logs`), коли `settings.detailed_logs_enabled=true` -- текстовий журнал статусу запиту в `log.txt` (опціонально/compat) -- необов'язкові глибокі журнали запитів/перекладів у `logs/`, коли `ENABLE_REQUEST_LOGS=true` -- кінцеві точки використання інформаційної панелі (`/api/usage/*`) для використання інтерфейсу користувача +## 5) Data Integrity -Детальне захоплення корисного навантаження запиту зберігає до чотирьох етапів корисного навантаження JSON на маршрутизований виклик: +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- необроблений запит, отриманий від клієнта -- перекладений запит, фактично надісланий вгору -- відповідь провайдера, реконструйована як JSON; потокові відповіді стискаються до остаточного підсумку та метаданих потоку -- остаточна відповідь клієнта, яку повертає OmniRoute; потокові відповіді зберігаються в тій самій компактній формі підсумку## Security-Sensitive Boundaries +## Observability and Operational Signals -- Секрет JWT (`JWT_SECRET`) захищає перевірку/підпис файлів cookie сеансу інструментальної панелі -- Початковий пароль початкового завантаження (`INITIAL_PASSWORD`) має бути явно налаштований для першого запуску. -- Ключ API HMAC Secret (`API_KEY_SECRET`) захищає згенерований локальний формат ключа API -- Секрети постачальника (ключі/токени API) зберігаються в локальній БД і повинні бути захищені на рівні файлової системи -- Кінцеві точки хмарної синхронізації покладаються на автентику ключа API + семантику ідентифікатора машини## Environment and Runtime Matrix +Runtime visibility sources: -Змінні середовища, які активно використовуються кодом: +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -- Додаток/автентифікація: `JWT_SECRET`, `INITIAL_PASSWORD` -- Зберігання: `DATA_DIR` -- Сумісна поведінка вузла: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Перевизначення додаткової бази зберігання (Linux/macOS, коли `DATA_DIR` не встановлено): `XDG_CONFIG_HOME` -- Хешування безпеки: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Ведення журналу: `ENABLE_REQUEST_LOGS` -- URL-адреси синхронізації/хмари: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- Вихідний проксі: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` і варіанти в нижньому регістрі -- Позначки функції SOCKS5: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- Допоміжні засоби платформи/виконання (не для конкретної програми): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +Detailed request payload capture stores up to four JSON payload stages per routed call: -1. `usageDb` і `localDb` спільно використовують ту саму базову політику каталогу (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) із міграцією старих файлів. -2. `/api/v1/route.ts` делегує той самий уніфікований конструктор каталогу, що використовується `/api/v1/models` (`src/app/api/v1/models/catalog.ts`), щоб уникнути семантичного дрейфу. -3. Реєстратор запитів записує повні заголовки/тіло, якщо ввімкнено; вважати каталог журналу конфіденційним. -4. Поведінка хмари залежить від правильності `NEXT_PUBLIC_BASE_URL` і доступності кінцевої точки хмари. -5. Каталог `open-sse/` публікується як `@omniroute/open-sse`**пакет робочої області npm**. Вихідний код імпортує його через `@omniroute/open-sse/...` (вирішено Next.js `transpilePackages`). Шляхи до файлів у цьому документі все ще використовують назву каталогу `open-sse/` для узгодженості. -6. Діаграми на інформаційній панелі використовують**Recharts**(на основі SVG) для доступної інтерактивної візуалізації аналітики (гістограми використання моделі, таблиці розбивки постачальників із показниками успіху). -7. Тести E2E використовують**Playwright**(`tests/e2e/`), запускають через `npm run test:e2e`. У модульних тестах використовується**Node.js тест-виконавець**(`tests/unit/`), запущений через `npm run test:unit`. Вихідним кодом у `src/` є**TypeScript**(`.ts`/`.tsx`); робоча область `open-sse/` залишається JavaScript (`.js`). -8. Сторінка налаштувань організована на 5 вкладках: Безпека, Маршрутизація (6 глобальних стратегій: спочатку заповнює, циклічна, p2c, випадкова, найменш використовувана, оптимізована за витратами), Стійкість (редаговані обмеження швидкості, автоматичний вимикач, політики), ШІ (бюджет мислення, системна підказка, кеш підказок), Додатково (проксі).## Operational Verification Checklist +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form -- Збірка з джерела: `npm run build` - — Зображення Docker: `docker build -t omniroute .` -- Запустіть службу та перевірте: +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: - `GET /api/settings` - `GET /api/v1/models` -- Цільова базова URL-адреса CLI має бути `http://:20128/v1`, коли `PORT=20128` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/uk-UA/docs/FEATURES.md b/docs/i18n/uk-UA/docs/FEATURES.md index 34e5958d3e..897f4b9134 100644 --- a/docs/i18n/uk-UA/docs/FEATURES.md +++ b/docs/i18n/uk-UA/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Візуальний путівник по кожному розділу інформаційної панелі OmniRoute.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Керуйте підключеннями постачальників AI: постачальників OAuth (Claude Code, Codex, Gemini CLI), постачальників ключів API (Groq, DeepSeek, OpenRouter) і безкоштовних постачальників (Qoder, Qwen, Kiro). Облікові записи Kiro включають відстеження кредитного балансу — залишок кредитів, загальну суму та дату поновлення можна побачити в Інформаційній панелі → Використання.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Створюйте комбіновані моделі маршрутизації за допомогою 6 стратегій: пріоритетної, зваженої, циклічної, випадкової, найменш використовуваної та економічно оптимізованої. Кожне комбо об’єднує кілька моделей із автоматичним резервним копіюванням і містить швидкі шаблони та перевірку готовності.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Комплексна аналітика використання із споживанням токенів, оцінками витрат, тепловими картами активності, тижневими діаграмами розподілу та розподілом за постачальниками.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Моніторинг у режимі реального часу: безвідмовна робота, пам’ять, версія, процентилі затримки (p50/p95/p99), статистика кешу та стани автоматичного вимикача постачальника.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Чотири режими для налагодження перекладів API:**Playground**(конвертер форматів),**Chat Tester**(живі запити),**Test Bench**(пакетні тести) і**Live Monitor**(потік у реальному часі).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Тестуйте будь-яку модель прямо з приладової панелі. Вибирайте провайдера, модель і кінцеву точку, пишіть підказки за допомогою Monaco Editor, транслюйте відповіді в режимі реального часу, переривайте посередині потоку та переглядайте показники часу.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Настроювані кольорові теми для всієї інформаційної панелі. Виберіть із 7 попередньо встановлених кольорів (кораловий, синій, червоний, зелений, фіолетовий, помаранчевий, блакитний) або створіть власну тему, вибравши будь-який шістнадцятковий колір. Підтримує світлий, темний і системний режими.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Комплексна панель налаштувань із вкладками: +Comprehensive settings panel with tabs: --**Загальні**— системне зберігання, керування резервним копіюванням (експорт/імпорт бази даних) -**Зовнішній вигляд**— селектор теми (темний/світлий/системний), попередньо встановлені теми кольорів і спеціальні кольори, видимість журналу стану здоров’я, елементи керування видимістю елементів бічної панелі -**Безпека**— захист кінцевих точок API, блокування настроюваних провайдерів, фільтрація IP-адрес, інформація про сеанс -**Маршрутизація**— Псевдоніми моделі, погіршення якості фонового завдання -**Стійкість**— постійне обмеження швидкості, налаштування автоматичного вимикача, автоматичне відключення заборонених облікових записів, моніторинг закінчення терміну дії постачальника -**Додатково**— заміна конфігурації, журнал аудиту конфігурації, резервний режим деградації![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Конфігурація в один клік для інструментів кодування ШІ: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor і Factory Droid. Функції автоматичного застосування/скидання конфігурації, профілів підключення та відображення моделі.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Інформаційна панель для виявлення та керування агентами CLI. Показує сітку з 14 вбудованих агентів (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) із: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Статус встановлення**— Встановлено/Не знайдено з визначенням версії -**Бейджи протоколу**— stdio, HTTP тощо. -**Користувацькі агенти**— реєструйте будь-який інструмент CLI за допомогою форми (ім’я, двійковий файл, команда версії, аргументи створення) -**Відповідність відбитків CLI**— перемикання для кожного постачальника, щоб відповідати власним підписам запиту CLI, зменшуючи ризик блокування, зберігаючи IP-адресу проксі-сервера--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Створюйте зображення, відео та музику з інформаційної панелі. Підтримує OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open і MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Реєстрація запитів у режимі реального часу з фільтрацією за постачальником, моделлю, обліковим записом і ключем API. Показує коди стану, використання маркерів, затримку та деталі відповіді.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Ваша уніфікована кінцева точка API з розбивкою можливостей: завершення чату, API відповідей, вбудовування, генерація зображень, переранжування, транскрипція аудіо, синтез мовлення з тексту, модерація та зареєстровані ключі API. Інтеграція Cloudflare Quick Tunnel і підтримка хмарного проксі для віддаленого доступу.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -Створення, область дії та відкликання ключів API. Кожен ключ можна обмежити певними моделями/постачальниками з повним доступом або дозволом лише на читання. Візуальне керування ключами з відстеженням використання.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Відстеження адміністративних дій із фільтрацією за типом дії, актором, метою, IP-адресою та міткою часу. Повна історія подій безпеки.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Власна настільна програма Electron для Windows, macOS і Linux. Запустіть OmniRoute як окрему програму з інтеграцією в системний трей, підтримкою в автономному режимі, автоматичним оновленням і встановленням одним клацанням миші. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Ключові особливості: +Key features: -- Опитування готовності сервера (без порожнього екрана при холодному запуску) - — Системний трей із керуванням портами -- Політика безпеки вмісту -- Одноразове блокування -- Автоматичне оновлення при перезавантаженні -- Інтерфейс користувача, залежний від платформи (світлофори macOS, панель заголовка за замовчуванням у Windows/Linux) -- Hardened Electron build packaging — символічні `node_modules` в автономному комплекті виявляються та відхиляються перед пакуванням, запобігаючи залежності часу виконання від комп’ютера збірки (версія 2.5.5+). +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Повну документацію див. [`electron/README.md`](../electron/README.md). +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/uk-UA/docs/TROUBLESHOOTING.md b/docs/i18n/uk-UA/docs/TROUBLESHOOTING.md index 116f720c25..a66e8440a7 100644 --- a/docs/i18n/uk-UA/docs/TROUBLESHOOTING.md +++ b/docs/i18n/uk-UA/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -Поширені проблеми та рішення для OmniRoute.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Проблема | Рішення | -| -------------------------------------------------------- | --------------------------------------------------------------------------- | --- | -| Перший вхід не працює | Установіть `INITIAL_PASSWORD` у `.env` (за умовчанням немає жорсткого коду) | -| Інформаційна панель відкривається на неправильному порту | Установіть `PORT=20128` і `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Немає журналів запитів у `logs/` | Установіть `ENABLE_REQUEST_LOGS=true` | -| EACCES: у дозволі відмовлено | Установіть `DATA_DIR=/path/to/writable/dir` на заміну `~/.omniroute` | -| Стратегія маршрутизації не зберігається | Оновлення до v1.4.11+ (виправлення схеми Zod для збереження налаштувань) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Причина:**Квота постачальника вичерпана. +**Cause:** Provider quota exhausted. -**Виправлення:** +**Fix:** -1. Перевірте трекер квот на інформаційній панелі -2. Використовуйте комбінацію з запасними рівнями -3. Перейдіть на дешевший/безкоштовний рівень### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Причина:**Квота підписки вичерпана. +### Rate Limiting -**Виправлення:** +**Cause:** Subscription quota exhausted. -- Додано запасний варіант: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Використовуйте GLM/MiniMax як дешеву резервну копію### OAuth Token Expired +**Fix:** -OmniRoute автоматично оновлює маркери. Якщо проблеми не зникають: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Інформаційна панель → Постачальник → Повторне підключення -2. Видаліть і повторно додайте підключення провайдера--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Переконайтеся, що `BASE_URL` вказує на ваш запущений екземпляр (наприклад, `http://localhost:20128`) -2. Переконайтеся, що `CLOUD_URL` вказує на кінцеву точку вашої хмари (наприклад, `https://omniroute.dev`) -3. Зберігайте значення `NEXT_PUBLIC_*` узгодженими зі значеннями на стороні сервера### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Проблема:**`Неочікуваний маркер 'd'...` на кінцевій точці хмари для непотокових викликів. +### Cloud `stream=false` Returns 500 -**Причина:**Upstream повертає корисне навантаження SSE, тоді як клієнт очікує JSON. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Обхідний шлях:**використовуйте `stream=true` для прямих дзвінків у хмарі. Місцеве середовище виконання включає SSE→JSON.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Створіть новий ключ із локальної інформаційної панелі (`/api/keys`) -2. Запустіть хмарну синхронізацію: увімкніть Cloud → Синхронізувати зараз -3. Старі/несинхронізовані ключі все ще можуть повертати «401» у хмарі--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Перевірте поля середовища виконання: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. Для портативного режиму: використовуйте цільове зображення `runner-cli` (в комплекті CLI) -3. Для режиму монтування хосту: встановіть `CLI_EXTRA_PATHS` і змонтуйте каталог bin хоста як лише для читання -4. Якщо `installed=true` і `runnable=false`: двійковий файл знайдено, але перевірка справності не пройшла### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Перевірте статистику використання в Інформаційна панель → Використання -2. Переключіть основну модель на GLM/MiniMax -3. Використовуйте безкоштовний рівень (Gemini CLI, Qoder) для некритичних завдань -4. Встановіть бюджет витрат на ключ API: Інформаційна панель → Ключі API → Бюджет--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Установіть `ENABLE_REQUEST_LOGS=true` у вашому файлі `.env`. Журнали відображаються в каталозі `logs/`.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Основний стан: `${DATA_DIR}/storage.sqlite` (постачальники, комбо, псевдоніми, ключі, налаштування) -- Використання: таблиці SQLite в `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + необов’язкові `${DATA_DIR}/log.txt` і `${DATA_DIR}/call_logs/` -- Журнали запитів: `/logs/...` (коли `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Коли автоматичний вимикач постачальника ВІДКРИТО, запити блокуються до закінчення часу відновлення. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Виправлення:** +**Fix:** -1. Перейдіть до**Інформаційна панель → Налаштування → Стійкість** -2. Перевірте плату автоматичного вимикача для постраждалого постачальника -3. Натисніть**Скинути все**, щоб очистити всі вимикачі, або зачекайте, доки закінчиться час відновлення -4. Перед скиданням переконайтеся, що постачальник дійсно доступний### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Якщо постачальник постійно переходить у стан ВІДКРИТО: +### Provider keeps tripping the circuit breaker -1. Перевірте**Інформаційна панель → Справність → Справність постачальника**на предмет шаблону збою -2. Перейдіть до**Налаштування → Стійкість → Профілі постачальників**і збільште поріг відмов -3. Перевірте, чи постачальник змінив обмеження API або вимагає повторної автентифікації -4. Перегляньте телеметрію затримки — висока затримка може спричинити збої, пов’язані з тайм-аутом--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Переконайтеся, що ви використовуєте правильний префікс: `deepgram/nova-3` або `assemblyai/best` -- Переконайтеся, що постачальник підключено в**Інформаційна панель → Постачальники**### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Перевірте підтримувані аудіоформати: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- Переконайтеся, що розмір файлу відповідає обмеженням постачальника (зазвичай < 25 МБ) -- Перевірте дійсність ключа API провайдера в картці провайдера--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Використовуйте**Інформаційну панель → Перекладач**, щоб усунути проблеми з перекладом формату: +Use **Dashboard → Translator** to debug format translation issues: -| Режим | Коли використовувати | -| ------------------------ | -------------------------------------------------------------------------------------------------------------- | ------------------------ | -| **Дитячий майданчик** | Порівняйте формати введення/виведення поруч — вставте невдалий запит, щоб побачити, як він перекладається | -| **Тестувальник чату** | Надсилайте живі повідомлення та перевіряйте повне корисне навантаження запитів/відповідей, включаючи заголовки | -| **Випробувальний стенд** | Запустіть пакетне тестування комбінацій форматів, щоб знайти, які переклади порушені | -| **Живий монітор** | Слідкуйте за потоком запитів у реальному часі, щоб виявити періодичні проблеми з перекладом | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Теги мислення не відображаються**— перевірте, чи підтримує цільовий постачальник мислення та налаштування бюджету мислення -**Відмова від викликів інструментів**— деякі переклади форматів можуть видаляти непідтримувані поля; перевірити в режимі Playground -**Відсутня системна підказка**— Клод і Близнюки по-різному обробляють системні підказки; перевірити результат перекладу -**SDK повертає необроблений рядок замість об’єкта**— Виправлено у версії 1.1.0: дезінфікуючий засіб відповіді тепер видаляє нестандартні поля (`x_groq`, `usage_breakdown` тощо), які спричиняють помилки перевірки OpenAI SDK Pydantic. -**GLM/ERNIE відхиляє `системну` роль**— Виправлено у версії 1.1.0: нормалізатор ролі автоматично об’єднує системні повідомлення в повідомлення користувача для несумісних моделей. -**`Роль розробника` не розпізнається**— Виправлено у версії 1.1.0: автоматично перетворено на `систему` для постачальників, які не є OpenAI -**`json_schema` не працює з Gemini**— Виправлено у версії 1.1.0: `response_format` тепер перетворено на `responseMimeType` + `responseSchema` Gemini.--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- Автоматичне обмеження швидкості стосується лише постачальників ключів API (не OAuth/підписки) -- Переконайтеся, що**Налаштування → Стійкість → Профілі постачальників**увімкнено автоматичне обмеження швидкості -- Перевірте, чи повертає постачальник коди статусу `429` або заголовки `Retry-After`### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -Профілі постачальників підтримують такі налаштування: +### Tuning exponential backoff --**Базова затримка**— початковий час очікування після першої помилки (за замовчуванням: 1 с) -**Макс. затримка**— обмеження максимального часу очікування (за замовчуванням: 30 с) -**Множник**— скільки збільшити затримку на послідовну помилку (за замовчуванням: 2x)### Anti-thundering herd +Provider profiles support these settings: -Коли багато одночасних запитів надходять до постачальника з обмеженою швидкістю, OmniRoute використовує м’ютекс + автоматичне обмеження швидкості для серіалізації запитів і запобігання каскадним помилкам. Це відбувається автоматично для постачальників ключів API.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Деякі користувачі OmniRoute розміщують шлюз перед стеками RAG або агентів. У цих налаштуваннях зазвичай можна побачити дивну модель: OmniRoute виглядає справним (провайдери працюють, профілі маршрутизації в порядку, немає сповіщень про обмеження швидкості), але остаточна відповідь все одно неправильна. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -На практиці ці інциденти зазвичай походять від нижнього трубопроводу RAG, а не від самого шлюзу. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Якщо вам потрібен спільний словник для опису цих збоїв, ви можете скористатися WFGY ProblemMap, зовнішнім текстовим ресурсом ліцензії MIT, який визначає шістнадцять повторюваних шаблонів збоїв RAG / LLM. На високому рівні він охоплює: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- дрейф пошуку та порушені межі контексту -- порожні або застарілі індекси та векторні сховища -- вбудовування проти семантичної невідповідності -- проблеми швидкої збірки та контекстного вікна -- зрив логіки та надто самовпевнені відповіді -- довгий ланцюг і збої координації агентів -- мультиагентна пам'ять і рольовий дрейф -- проблеми з розгортанням і завантаженням +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -Ідея проста: +The idea is simple: -1. Коли ви досліджуєте погану відповідь, зафіксуйте: - - завдання та запит користувача - - комбінація маршрутів або провайдерів у OmniRoute - - будь-який контекст RAG, що використовується нижче за потоком (отримані документи, виклики інструментів тощо) -2. Зіставте інцидент з одним або двома номерами WFGY ProblemMap («No.1» … `No.16`). -3. Збережіть номер на власній інформаційній панелі, Runbook або системі відстеження подій поруч із журналами OmniRoute. -4. Використовуйте відповідну сторінку WFGY, щоб вирішити, чи потрібно вам змінити стек RAG, стратегію отримання чи маршрутизації. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -Повний текст і конкретні рецепти доступні тут (ліцензія MIT, лише текст): +Full text and concrete recipes live here (MIT license, text only): [WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -Ви можете проігнорувати цей розділ, якщо ви не запускаєте конвеєри RAG або агента за OmniRoute.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**Проблеми GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Архітектура**: див. [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) для внутрішніх деталей -**Довідник API**: див. [`docs/API_REFERENCE.md`](API_REFERENCE.md) для всіх кінцевих точок -**Інформаційна панель справності**: перевірте**Інформаційна панель → Здоров’я**, щоб дізнатися про стан системи в реальному часі -**Перекладач**: використовуйте**Інформаційна панель → Перекладач**для усунення проблем із форматом +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/uk-UA/llm.txt b/docs/i18n/uk-UA/llm.txt new file mode 100644 index 0000000000..9f0d90a075 --- /dev/null +++ b/docs/i18n/uk-UA/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Українська) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Огляд + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Безпека +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/vi/README.md b/docs/i18n/vi/README.md index 563dc0e4e1..a71bc3f9c2 100644 --- a/docs/i18n/vi/README.md +++ b/docs/i18n/vi/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_Proxy API phổ quát của bạn — một điểm cuối, hơn 60 nhà cung cấp, không có thời gian ngừng hoạt động. Hiện có**Máy chủ MCP (25 công cụ)**,**Giao thức A2A**,**Hệ thống bộ nhớ/kỹ năng**&**Ứng dụng máy tính để bàn điện tử**._ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**Hoàn thành cuộc trò chuyện • Nhúng · Tạo hình ảnh · Video · Âm nhạc · Âm thanh · Sắp xếp lại ·**Tìm kiếm trên web**· Máy chủ MCP · Giao thức A2A · 100% TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _Proxy API phổ quát của bạn — một điểm cuối, hơn 60 nhà cung c [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 Trang web](https://omniroute.online) • [🚀 Bắt đầu nhanh](#-quick-start) • [💡 Tính năng](#-key-features) • [📖 Tài liệu](#-tài liệu) • [💰 Giá](#-giá trong nháy mắt) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**Có sẵn trong:**🇺🇸 [Tiếng Anh](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Tiếng Tây Ban Nha](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [tiếng Ý](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Tiếng Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Tiếng Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Hà Lan](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Bồ Đào Nha)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Philippines](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,554 +60,629 @@ _Proxy API phổ quát của bạn — một điểm cuối, hơn 60 nhà cung c ## 📸 Dashboard Preview - +
+Click to see dashboard screenshots -Nhấp để xem ảnh chụp màn hình trang tổng quan +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | -| Trang | Ảnh chụp màn hình | -| ------------------- | -------------------------------------------------- | ---------- | -| **Nhà cung cấp** | ![Nhà cung cấp](docs/screenshots/01-providers.png) | -| **Combo** | ![Combos](docs/screenshots/02-combos.png) | -| **Phân tích** | ![Analytics](docs/screenshots/03-analytics.png) | -| **Sức khỏe** | ![Sức khỏe](docs/screenshots/04-health.png) | -| **Người dịch** | ![Translator](docs/screenshots/05-translator.png) | -| **Cài đặt** | ![Cài đặt](docs/screenshots/06-settings.png) | -| **Công cụ CLI** | ![Công cụ CLI](docs/screenshots/07-cli-tools.png) | -| **Nhật ký sử dụng** | ![Cách sử dụng](docs/screenshots/08-usage.png) | -| **Điểm cuối** | ![Điểm cuối](docs/screenshots/09-endpoint.png) |
| + --- ### 🤖 Free AI Provider for your favorite coding agents -_Kết nối mọi công cụ IDE hoặc CLI được hỗ trợ bởi AI thông qua OmniRoute — cổng API miễn phí để mã hóa không giới hạn._ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ - - - - -OpenClaw
-OpenClaw -

-⭐ 205K - - - -NanoBot
-NanoBot -

-⭐ 20,9K - - - -PicoClaw
-PicoClaw -

-⭐ 14,6K - - - -ZeroClaw
-ZeroClaw -

-⭐ 9,9K - - - -IronClaw
-Móng vuốt sắt -

-⭐ 2.1K - - - - - -OpenCode
-Mã mở -

-⭐ 106K - - - -Codex CLI
-Codex CLI -

-⭐ 60,8K - - - -Mã Claude
-Mã Claude -

-⭐ 67,3K - - - -Gemini CLI
-Song Tử CLI -

-⭐ 94,7K - - - -Mã Kilo
-Mã Kilo -

-⭐ 15,5K - - -
+ + + + + + + + + + + + + + + +
+ + OpenClaw
+ OpenClaw +

+ ⭐ 205K +
+ + NanoBot
+ NanoBot +

+ ⭐ 20.9K +
+ + PicoClaw
+ PicoClaw +

+ ⭐ 14.6K +
+ + ZeroClaw
+ ZeroClaw +

+ ⭐ 9.9K +
+ + IronClaw
+ IronClaw +

+ ⭐ 2.1K +
+ + OpenCode
+ OpenCode +

+ ⭐ 106K +
+ + Codex CLI
+ Codex CLI +

+ ⭐ 60.8K +
+ + Claude Code
+ Claude Code +

+ ⭐ 67.3K +
+ + Gemini CLI
+ Gemini CLI +

+ ⭐ 94.7K +
+ + Kilo Code
+ Kilo Code +

+ ⭐ 15.5K +
-📡 Tất cả đại lý kết nối qua http://localhost:20128/v1 hoặc http://cloud.omniroute.online/v1 — một cấu hình, mô hình và hạn ngạch không giới hạn--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**Ngưng lãng phí tiền và đạt đến giới hạn:** +**Stop wasting money and hitting limits:** -- Hạn mức đăng ký hết hạn không được sử dụng hàng tháng -- Giới hạn tốc độ ngăn bạn viết mã giữa chừng -- API đắt ($20-50/tháng cho mỗi nhà cung cấp) -- Chuyển đổi thủ công giữa các nhà cung cấp +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute giải quyết vấn đề này:** +**OmniRoute solves this:** -- ✅**Tối đa hóa số lượt đăng ký**- Theo dõi hạn ngạch, sử dụng từng bit trước khi đặt lại -- ✅**Tự động dự phòng**- Đăng ký → Khóa API → Giá rẻ → Miễn phí, không có thời gian ngừng hoạt động -- ✅**Nhiều tài khoản**- Luân chuyển giữa các tài khoản cho mỗi nhà cung cấp -- ✅**Universal**- Hoạt động với Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, mọi công cụ CLI--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**Tham gia cộng đồng của chúng tôi!**[Nhóm WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Nhận trợ giúp, chia sẻ mẹo và luôn cập nhật. +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**Trang web**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Vấn đề**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Nhóm cộng đồng](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Đóng góp**: Xem [CONTRIBUTING.md](CONTRIBUTING.md), mở PR hoặc chọn `ấn bản đầu tiên hay` -**Dự án gốc**: [9router của decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -Khi mở một vấn đề, vui lòng chạy lệnh system-info và đính kèm tệp được tạo:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -Điều này tạo ra một `system-info.txt` với phiên bản Node.js, phiên bản OmniRoute, chi tiết hệ điều hành, các công cụ CLI đã cài đặt (qoder, gemini, claude, codex, antiGravity, droid, v.v.), trạng thái Docker/PM2 và các gói hệ thống — mọi thứ chúng tôi cần để tái hiện vấn đề của bạn một cách nhanh chóng. Đính kèm tệp trực tiếp vào vấn đề GitHub của bạn.--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**Mọi nhà phát triển sử dụng công cụ AI đều phải đối mặt với những vấn đề này hàng ngày.**OmniRoute được xây dựng để giải quyết tất cả — từ chi phí vượt mức cho đến chặn khu vực, từ luồng OAuth bị hỏng đến hoạt động giao thức và khả năng quan sát của doanh nghiệp. +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. - -💸 1. "Tôi trả tiền cho một thuê bao đắt tiền nhưng vẫn bị gián đoạn bởi các giới hạn"
+
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -Các nhà phát triển trả 20–200 USD/tháng cho Claude Pro, Codex Pro hoặc GitHub Copilot. Ngay cả khi trả tiền, hạn ngạch vẫn có mức trần - 5 giờ sử dụng, giới hạn hàng tuần hoặc giới hạn tốc độ mỗi phút. Giữa phiên mã hóa, nhà cung cấp ngừng phản hồi và nhà phát triển mất đi dòng chảy và năng suất. +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**Cách OmniRoute giải quyết vấn đề này:** +**How OmniRoute solves it:** --**Dự phòng 4 tầng thông minh**— Nếu hết hạn ngạch đăng ký, tự động chuyển hướng đến Khóa API → Giá rẻ → Miễn phí mà không cần can thiệp thủ công --**Theo dõi giới hạn của nhà cung cấp**— Ảnh chụp nhanh hạn ngạch được lưu trong bộ nhớ đệm làm mới theo lịch trình phía máy chủ (mặc định `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) với tính năng làm mới thủ công có sẵn trong giao diện người dùng --**Hỗ trợ nhiều tài khoản**— Nhiều tài khoản cho mỗi nhà cung cấp với tính năng tự động quay vòng — khi hết một tài khoản, hãy chuyển sang tài khoản tiếp theo --**Combo tùy chỉnh**— Chuỗi dự phòng có thể tùy chỉnh với 9 chiến lược cân bằng (ưu tiên, có trọng số, điền trước, quay vòng, P2C, ngẫu nhiên, ít sử dụng nhất, tối ưu hóa chi phí, ngẫu nhiên nghiêm ngặt) --**Hạn ngạch kinh doanh Codex**— Giám sát hạn ngạch không gian làm việc của Doanh nghiệp/Nhóm trực tiếp trong bảng điều khiển
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard - -🔌 2. "Tôi cần sử dụng nhiều nhà cung cấp nhưng mỗi nhà cung cấp có một API khác nhau" + -OpenAI sử dụng một định dạng, Claude (Anthropic) sử dụng một định dạng khác, Gemini lại sử dụng một định dạng khác. Nếu nhà phát triển muốn thử nghiệm các mô hình từ các nhà cung cấp khác nhau hoặc dự phòng giữa các nhà cung cấp đó, họ cần phải định cấu hình lại SDK, thay đổi điểm cuối, xử lý các định dạng không tương thích. Các nhà cung cấp tùy chỉnh (FriendLI, NIM) có các điểm cuối mô hình không chuẩn. +
+🔌 2. "I need to use multiple providers but each has a different API" -**Cách OmniRoute giải quyết vấn đề này:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**Điểm cuối hợp nhất**— Một `http://localhost:20128/v1` duy nhất đóng vai trò là proxy cho tất cả hơn 60 nhà cung cấp --**Dịch định dạng**— Tự động và minh bạch: OpenAI ↔ Claude ↔ Gemini ↔ API phản hồi --**Sạch hóa phản hồi**— Loại bỏ các trường không chuẩn (`x_groq`, `usage_breakdown`, `service_tier`) phá vỡ OpenAI SDK v1.83+ --**Chuẩn hóa vai trò**— Chuyển đổi `nhà phát triển` → `hệ thống` cho các nhà cung cấp không phải OpenAI; `system` → `user` cho GLM/ERNIE --**Trích xuất thẻ suy nghĩ**— Trích xuất các khối `` từ các mô hình như DeepSeek R1 thành `reasoning_content` được tiêu chuẩn hóa --**Đầu ra có cấu trúc cho Gemini**— `json_schema` → `responseMimeType`/`responseSchema` chuyển đổi tự động --**`stream` mặc định là `false`**— Căn chỉnh với thông số OpenAI, tránh SSE không mong muốn trong SDK Python/Rust/Go
+**How OmniRoute solves it:** - -🌐 3. "Nhà cung cấp AI của tôi chặn khu vực/quốc gia của tôi" +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -Các nhà cung cấp như OpenAI/Codex chặn quyền truy cập từ các khu vực địa lý nhất định. Người dùng gặp phải các lỗi như `unsupported_country_zone_territory` trong quá trình kết nối OAuth và API. Điều này đặc biệt gây khó chịu cho các nhà phát triển từ các nước đang phát triển. + -**Cách OmniRoute giải quyết vấn đề này:** +
+🌐 3. "My AI provider blocks my region/country" --**Cấu hình proxy 3 cấp**— Proxy có thể định cấu hình ở 3 cấp độ: toàn cầu (tất cả lưu lượng truy cập), mỗi nhà cung cấp (chỉ một nhà cung cấp) và mỗi kết nối/khóa --**Huy hiệu proxy được mã hóa màu**— Chỉ báo trực quan: 🟢 proxy toàn cầu, 🟡 proxy nhà cung cấp, 🔵 proxy kết nối, luôn hiển thị IP --**Trao đổi mã thông báo OAuth thông qua proxy**— Luồng OAuth cũng đi qua proxy, giải quyết `unsupported_country_zone_territory` --**Kiểm tra kết nối qua Proxy**— Kiểm tra kết nối sử dụng proxy đã định cấu hình (không cần bỏ qua trực tiếp nữa) --**Hỗ trợ SOCKS5**— Hỗ trợ proxy SOCKS5 đầy đủ cho định tuyến đi --**Giả mạo dấu vân tay TLS**— Dấu vân tay TLS giống trình duyệt thông qua `wreq-js` để vượt qua khả năng phát hiện bot --**🔏 Khớp dấu vân tay CLI**— Sắp xếp lại các tiêu đề và trường nội dung để khớp với chữ ký nhị phân CLI gốc, giảm đáng kể rủi ro gắn cờ tài khoản. IP proxy được giữ nguyên — bạn có được cả mặt nạ IP**và**ẩn cùng lúc
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. - -🆓 4. "Tôi muốn sử dụng AI để viết mã nhưng tôi không có tiền" +**How OmniRoute solves it:** -Không phải ai cũng có thể trả 20–200 USD/tháng để đăng ký AI. Sinh viên, nhà phát triển từ các quốc gia mới nổi, những người có sở thích và người làm việc tự do cần được tiếp cận với các mô hình chất lượng với chi phí bằng 0. +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**Cách OmniRoute giải quyết vấn đề này:** + --**Tích hợp sẵn nhà cung cấp cấp miễn phí**— Hỗ trợ riêng cho nhà cung cấp miễn phí 100%: Qoder (5 mô hình không giới hạn qua OAuth: kimi-k2-thinking, qwen3-code-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 mô hình không giới hạn: qwen3-code-plus, qwen3-code-flash, qwen3-code-next, Vision-model), Kiro (Claude + AWS Builder ID miễn phí), Gemini CLI (miễn phí 180K token/tháng) --**Ollama Cloud**— Các mô hình Ollama được lưu trữ trên đám mây tại `api.ollama.com` với bậc "Sử dụng nhẹ" miễn phí; sử dụng tiền tố `ollamacloud/` --**Combo chỉ miễn phí**— Chuỗi `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-code-plus` = $0/tháng mà không có thời gian ngừng hoạt động --**Truy cập miễn phí NVIDIA NIM**— ~40 RPM dành cho nhà phát triển - truy cập miễn phí vĩnh viễn vào hơn 70 mẫu tại build.nvidia.com (chuyển từ tín dụng sang giới hạn tỷ lệ thuần túy) --**Chiến lược tối ưu hóa chi phí**— Chiến lược định tuyến tự động chọn nhà cung cấp sẵn có rẻ nhất +
+🆓 4. "I want to use AI for coding but I have no money" - -🔒 5. "Tôi cần bảo vệ cổng AI của mình khỏi bị truy cập trái phép" +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -Khi đưa cổng AI vào mạng (LAN, VPS, Docker), bất kỳ ai có địa chỉ đều có thể sử dụng mã thông báo/hạn ngạch của nhà phát triển. Nếu không có biện pháp bảo vệ, các API dễ bị lạm dụng, chèn ép và lạm dụng. +**How OmniRoute solves it:** -**Cách OmniRoute giải quyết vấn đề này:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**Quản lý khóa API**— Tạo, xoay vòng và xác định phạm vi cho mỗi nhà cung cấp với trang `/dashboard/api-manager` chuyên dụng --**Quyền cấp mô hình**— Hạn chế khóa API đối với các mô hình cụ thể (`openai/*`, mẫu ký tự đại diện), với nút chuyển đổi Cho phép tất cả/Hạn chế --**Bảo vệ điểm cuối API**— Yêu cầu khóa cho `/v1/models` và chặn các nhà cung cấp cụ thể khỏi danh sách --**Auth Guard + CSRF Protection**— Tất cả các tuyến bảng điều khiển được bảo vệ bằng phần mềm trung gian `withAuth` + mã thông báo CSRF --**Giới hạn tốc độ**— Giới hạn tốc độ trên mỗi IP với các cửa sổ có thể định cấu hình --**Lọc IP**— Danh sách cho phép/danh sách chặn để kiểm soát truy cập --**Prompt Tiêm Guard**— Khử trùng các mẫu nhắc nhở độc hại --**Mã hóa AES-256-GCM**— Thông tin xác thực được mã hóa ở trạng thái lưu trữ
+ - -🛑 6. "Nhà cung cấp của tôi ngừng hoạt động và tôi mất luồng mã hóa" +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -Các nhà cung cấp AI có thể trở nên không ổn định, trả về lỗi 5xx hoặc đạt giới hạn tốc độ tạm thời. Nếu một nhà phát triển phụ thuộc vào một nhà cung cấp duy nhất thì họ sẽ bị gián đoạn. Nếu không có bộ ngắt mạch, việc thử lại nhiều lần có thể làm hỏng ứng dụng. +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**Cách OmniRoute giải quyết vấn đề này:** +**How OmniRoute solves it:** --**Bộ ngắt mạch trên mỗi mô hình**— Tự động mở/đóng với các ngưỡng và thời gian hồi chiêu có thể định cấu hình (Đóng/Mở/Nửa mở), trong phạm vi mỗi mô hình để tránh xếp tầng --**Thời gian chờ theo cấp số nhân**— Độ trễ thử lại lũy tiến --**Bầy chống sấm sét**— Mutex + bảo vệ semaphore chống lại các cơn bão thử lại đồng thời --**Chuỗi dự phòng kết hợp**— Nếu nhà cung cấp chính không thành công, nó sẽ tự động rơi qua chuỗi mà không cần can thiệp --**Combo Circuit Breaker**— Tự động vô hiệu hóa các nhà cung cấp bị lỗi trong chuỗi kết hợp --**Bảng thông tin sức khỏe**— Giám sát thời gian hoạt động, trạng thái ngắt mạch, khóa, số liệu thống kê bộ nhớ đệm, độ trễ p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest - -🔧 7. "Cấu hình từng công cụ AI thật tẻ nhạt và lặp đi lặp lại" + -Nhà phát triển sử dụng Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Mỗi công cụ cần một cấu hình khác nhau (điểm cuối API, khóa, mô hình). Việc cấu hình lại khi chuyển đổi nhà cung cấp hoặc mô hình là một sự lãng phí thời gian. +
+🛑 6. "My provider went down and I lost my coding flow" -**Cách OmniRoute giải quyết vấn đề này:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**Bảng điều khiển công cụ CLI**— Trang chuyên dụng với thiết lập bằng một cú nhấp chuột cho Claude Code, Codex CLI, OpenClaw, Kilo Code, AntiGravity, Cline --**Trình tạo cấu hình GitHub Copilot**— Tạo `chatLanguageModels.json` cho Mã VS với lựa chọn mô hình hàng loạt --**Trình hướng dẫn tích hợp**— Thiết lập 4 bước có hướng dẫn cho người dùng lần đầu --**Một điểm cuối, tất cả các mô hình**— Định cấu hình `http://localhost:20128/v1` một lần, truy cập hơn 60 nhà cung cấp
+**How OmniRoute solves it:** - -🔑 8. "Quản lý mã thông báo OAuth từ nhiều nhà cung cấp là địa ngục" +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Các nhà phát triển cần phải xác thực lại liên tục, xử lý lỗi `client_secret bị thiếu`, `redirect_uri_mismatch` và các lỗi trên máy chủ từ xa. OAuth trên LAN/VPS đặc biệt có vấn đề. + -**Cách OmniRoute giải quyết vấn đề này:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**Tự động làm mới mã thông báo**— Làm mới mã thông báo OAuth ở chế độ nền trước khi hết hạn --**Tích hợp OAuth 2.0 (PKCE)**— Luồng tự động cho Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder --**OAuth nhiều tài khoản**— Nhiều tài khoản cho mỗi nhà cung cấp thông qua trích xuất mã thông báo JWT/ID --**OAuth LAN/Remote Fix**— Phát hiện IP riêng cho `redirect_uri` + chế độ URL thủ công cho máy chủ từ xa --**OAuth đằng sau Nginx**— Sử dụng `window.location.origin` để tương thích với proxy ngược --**Hướng dẫn OAuth từ xa**— Hướng dẫn từng bước về thông tin đăng nhập Google Cloud trên VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. - -📊 9. "Tôi không biết mình đang chi bao nhiêu và ở đâu" +**How OmniRoute solves it:** -Các nhà phát triển sử dụng nhiều nhà cung cấp trả phí nhưng không có quan điểm thống nhất về chi tiêu. Mỗi nhà cung cấp có trang tổng quan thanh toán riêng nhưng không có chế độ xem tổng hợp. Chi phí bất ngờ có thể chồng chất. +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**Cách OmniRoute giải quyết vấn đề này:** + --**Bảng thông tin phân tích chi phí**— Theo dõi chi phí mỗi mã thông báo và quản lý ngân sách cho mỗi nhà cung cấp --**Giới hạn ngân sách cho mỗi cấp**— Mức chi tiêu trần cho mỗi cấp kích hoạt dự phòng tự động --**Cấu hình định giá theo mẫu**— Giá có thể định cấu hình cho mỗi mẫu --**Thống kê sử dụng trên mỗi khóa API**— Số lượng yêu cầu và dấu thời gian được sử dụng lần cuối trên mỗi khóa --**Bảng thông tin phân tích**— Thẻ thống kê, biểu đồ sử dụng mô hình, bảng nhà cung cấp với tỷ lệ thành công và độ trễ +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" - -🐛 10. "Tôi không thể chẩn đoán lỗi và sự cố trong cuộc gọi AI" +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -Khi cuộc gọi không thành công, nhà phát triển không biết liệu đó có phải là giới hạn tốc độ, mã thông báo đã hết hạn, sai định dạng hay lỗi nhà cung cấp hay không. Nhật ký bị phân mảnh trên các thiết bị đầu cuối khác nhau. Nếu không có khả năng quan sát được thì việc gỡ lỗi chỉ là thử và sai. +**How OmniRoute solves it:** -**Cách OmniRoute giải quyết vấn đề này:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**Bảng điều khiển nhật ký hợp nhất**— 4 tab: Nhật ký yêu cầu, Nhật ký proxy, Nhật ký kiểm tra, Bảng điều khiển --**Trình xem nhật ký bảng điều khiển**— Trình xem kiểu thiết bị đầu cuối thời gian thực với các cấp độ được mã hóa màu, tự động cuộn, tìm kiếm, lọc --**Nhật ký proxy SQLite**— Nhật ký liên tục vẫn tồn tại khi máy chủ khởi động lại --**Sân chơi dịch thuật**— 4 chế độ gỡ lỗi: Sân chơi (dịch định dạng), Trình kiểm tra trò chuyện (khứ hồi), Bàn thử nghiệm (hàng loạt), Giám sát trực tiếp (thời gian thực) --**Yêu cầu đo từ xa**— độ trễ p50/p95/p99 + truy tìm X-Request-Id --**Ghi nhật ký dựa trên tệp có xoay vòng**— Nhật ký ứng dụng xoay vòng theo kích thước, ngày lưu giữ và số lượng lưu trữ; các tạo phẩm trong nhật ký cuộc gọi xoay vòng theo số ngày lưu giữ và số lượng tệp --**Báo cáo thông tin hệ thống**— `npm run system-info` tạo `system-info.txt` với môi trường đầy đủ của bạn (Phiên bản nút, phiên bản OmniRoute, HĐH, công cụ CLI, trạng thái Docker/PM2). Đính kèm nó khi báo cáo vấn đề để phân loại ngay lập tức.
+ - -🏗️ 11. "Triển khai và bảo trì cổng rất phức tạp" +
+📊 9. "I don't know how much I'm spending or where" -Việc cài đặt, định cấu hình và duy trì proxy AI trên các môi trường khác nhau (cục bộ, VPS, Docker, đám mây) tốn nhiều công sức. Các vấn đề như đường dẫn được mã hóa cứng, `EACCES` trên thư mục, xung đột cổng và các bản dựng đa nền tảng sẽ gây thêm rắc rối. +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**Cách OmniRoute giải quyết vấn đề này:** +**How OmniRoute solves it:** --**npm cài đặt toàn cầu**— `npm cài đặt -g omniroute && omniroute` — xong --**Docker Đa nền tảng**— AMD64 + ARM64 gốc (Apple Silicon, AWS Graviton, Raspberry Pi) --**Docker Compose Profiles**— `base` (không có công cụ CLI) và `cli` (với Claude Code, Codex, OpenClaw) --**Ứng dụng máy tính để bàn điện tử**— Ứng dụng gốc dành cho Windows/macOS/Linux có khay hệ thống, tự động khởi động, chế độ ngoại tuyến --**Chế độ chia cổng**— API và Bảng điều khiển trên các cổng riêng biệt cho các tình huống nâng cao (proxy ngược, mạng vùng chứa) --**Cloud Sync**— Đồng bộ hóa cấu hình giữa các thiết bị thông qua Cloudflare Workers --**Sao lưu DB**— Tự động sao lưu, khôi phục, xuất và nhập tất cả cài đặt, với `DISABLE_SQLITE_AUTO_BACKUP` để sao lưu được quản lý bên ngoài
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency - -🌍 12. "Giao diện chỉ có tiếng Anh và nhóm của tôi không nói được tiếng Anh" + -Các đội ở các quốc gia không nói tiếng Anh, đặc biệt là ở Châu Mỹ Latinh, Châu Á và Châu Âu, gặp khó khăn với giao diện chỉ có tiếng Anh. Rào cản ngôn ngữ làm giảm khả năng tiếp nhận và tăng lỗi cấu hình. +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**Cách OmniRoute giải quyết vấn đề này:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**Bảng điều khiển i18n — 30 ngôn ngữ**— Tất cả hơn 500 phím được dịch bao gồm tiếng Ả Rập, tiếng Bungari, tiếng Đan Mạch, tiếng Đức, tiếng Tây Ban Nha, tiếng Phần Lan, tiếng Pháp, tiếng Do Thái, tiếng Hindi, tiếng Hungary, tiếng Indonesia, tiếng Ý, tiếng Nhật, tiếng Hàn, tiếng Mã Lai, tiếng Hà Lan, tiếng Na Uy, tiếng Ba Lan, tiếng Bồ Đào Nha (PT/BR), tiếng Rumani, tiếng Nga, tiếng Slovak, tiếng Thụy Điển, tiếng Thái, tiếng Ukraina, tiếng Việt, tiếng Trung, tiếng Philipin, tiếng Anh --**Hỗ trợ RTL**— Hỗ trợ từ phải sang trái cho tiếng Ả Rập và tiếng Do Thái --**README đa ngôn ngữ**— 30 bản dịch tài liệu hoàn chỉnh --**Bộ chọn ngôn ngữ**— Biểu tượng quả cầu trong tiêu đề để chuyển đổi theo thời gian thực
+**How OmniRoute solves it:** - -🔄 13. "Tôi cần nhiều hơn là trò chuyện — tôi cần nội dung nhúng, hình ảnh, âm thanh" +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -AI không chỉ hoàn thành cuộc trò chuyện. Nhà phát triển cần tạo hình ảnh, phiên âm âm thanh, tạo phần nhúng cho RAG, sắp xếp lại tài liệu và kiểm duyệt nội dung. Mỗi API có điểm cuối và định dạng khác nhau. + -**Cách OmniRoute giải quyết vấn đề này:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Embeddings**— `/v1/embeddings` với 6 nhà cung cấp và hơn 9 mô hình --**Tạo hình ảnh**— `/v1/images/thế hệ` với 10 nhà cung cấp và hơn 20 mô hình (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, AntiGravity, SD WebUI, ComfyUI) --**Chuyển văn bản thành video**— `/v1/video/thế hệ` — ComfyUI (AnimateDiff, SVD) và SD WebUI --**Chuyển văn bản thành nhạc**— `/v1/music/thế hệ` — ComfyUI (Mở âm thanh ổn định, MusicGen) --**Phiên âm âm thanh**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 --**Chuyển văn bản thành giọng nói**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + các nhà cung cấp hiện có --**Kiểm duyệt**— `/v1/moderations` — Kiểm tra an toàn nội dung --**Sắp xếp lại**— `/v1/rerank` — Sắp xếp lại mức độ liên quan của tài liệu --**API phản hồi**— Hỗ trợ đầy đủ `/v1/responses` cho Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. - +**How OmniRoute solves it:** + +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups + + + +
+🌍 12. "The interface is English-only and my team doesn't speak English" + +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. + +**How OmniRoute solves it:** + +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching + +
+ +
+🔄 13. "I need more than chat — I need embeddings, images, audio" + +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. + +**How OmniRoute solves it:** + +- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex + +
+ +
🧪 14. "I have no way to test and compare quality across models" -Các nhà phát triển muốn biết mô hình nào là tốt nhất cho trường hợp sử dụng của họ — mã, dịch thuật, lý luận — nhưng việc so sánh thủ công rất chậm. Không có công cụ đánh giá tích hợp nào tồn tại. +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -**Cách OmniRoute giải quyết vấn đề này:** +**How OmniRoute solves it:** --**Đánh giá LLM**— Bộ thử nghiệm vàng với 10 trường hợp tải sẵn bao gồm lời chào, toán, địa lý, tạo mã, tuân thủ JSON, dịch thuật, đánh dấu, từ chối an toàn --**4 Chiến lược kết hợp**— `chính xác`, `contains`, `regex`, `custom` (hàm JS) --**Băng thử nghiệm sân chơi dịch giả**— Thử nghiệm hàng loạt với nhiều đầu vào và đầu ra dự kiến, so sánh giữa các nhà cung cấp --**Trình kiểm tra trò chuyện**— Toàn bộ chuyến đi với kết xuất phản hồi trực quan --**Live Monitor**— Luồng thời gian thực của tất cả các yêu cầu truyền qua proxy
+- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy - -📈 15. "Tôi cần mở rộng quy mô mà không làm giảm hiệu suất" + -Khi khối lượng yêu cầu tăng lên mà không lưu vào bộ nhớ đệm thì các câu hỏi tương tự sẽ tạo ra chi phí trùng lặp. Nếu không có tính tạm thời, các yêu cầu trùng lặp sẽ bị lãng phí. Giới hạn tỷ lệ cho mỗi nhà cung cấp phải được tôn trọng. +
+📈 15. "I need to scale without losing performance" -**Cách OmniRoute giải quyết vấn đề này:** +As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. --**Bộ nhớ đệm ngữ nghĩa**— Bộ nhớ đệm hai tầng (chữ ký + ngữ nghĩa) giúp giảm chi phí và độ trễ --**Yêu cầu Idempotency**— Khoảng thời gian loại bỏ trùng lặp 5 giây cho các yêu cầu giống hệt nhau --**Phát hiện giới hạn tỷ lệ**— RPM của mỗi nhà cung cấp, khoảng cách tối thiểu và theo dõi đồng thời tối đa --**Giới hạn tỷ lệ có thể chỉnh sửa**— Giá trị mặc định có thể định cấu hình trong Cài đặt → Khả năng phục hồi với tính bền bỉ --**Bộ đệm xác thực khóa API**— Bộ đệm 3 tầng cho hiệu suất sản xuất --**Bảng thông tin sức khỏe với phép đo từ xa**— độ trễ p50/p95/p99, số liệu thống kê bộ nhớ đệm, thời gian hoạt động
+**How OmniRoute solves it:** - -🤖 16. "Tôi muốn kiểm soát hành vi kiểu mẫu trên toàn cầu" +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -Các nhà phát triển muốn tất cả phản hồi bằng một ngôn ngữ cụ thể, với giọng điệu cụ thể hoặc muốn giới hạn các mã thông báo lý luận. Việc định cấu hình điều này trong mọi công cụ/yêu cầu là không thực tế. + -**Cách OmniRoute giải quyết vấn đề này:** +
+🤖 16. "I want to control model behavior globally" --**Tiêm nhắc nhở hệ thống**— Lời nhắc chung được áp dụng cho tất cả các yêu cầu --**Xác thực ngân sách tư duy**— Kiểm soát phân bổ mã thông báo hợp lý cho mỗi yêu cầu (chuyển qua, tự động, tùy chỉnh, thích ứng) --**9 Chiến lược định tuyến**— Chiến lược toàn cầu xác định cách phân phối yêu cầu --**Bộ định tuyến ký tự đại diện**— mẫu `nhà cung cấp/*` định tuyến động tới bất kỳ nhà cung cấp nào --**Bật/Tắt kết hợp Chuyển đổi**— Chuyển đổi kết hợp trực tiếp từ bảng điều khiển --**Chuyển đổi nhà cung cấp**— Bật/tắt tất cả kết nối cho nhà cung cấp chỉ bằng một cú nhấp chuột --**Nhà cung cấp bị chặn**— Loại trừ các nhà cung cấp cụ thể khỏi danh sách `/v1/models`
+Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. - -🧰 17. "Tôi cần các công cụ MCP như khả năng sản phẩm hạng nhất" +**How OmniRoute solves it:** -Nhiều cổng AI chỉ hiển thị MCP dưới dạng chi tiết triển khai ẩn. Các nhóm cần một lớp hoạt động rõ ràng và dễ quản lý. +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -**Cách OmniRoute giải quyết vấn đề này:** + -- MCP xuất hiện trong tab điều hướng bảng điều khiển và giao thức điểm cuối -- Trang quản lý MCP chuyên dụng với quy trình, công cụ, phạm vi và kiểm tra -- Tích hợp tính năng khởi động nhanh cho `omniroute --mcp` và cài đặt ứng dụng khách +
+🧰 17. "I need MCP tools as first-class product capabilities" - -🧠 18. "Tôi cần phối hợp A2A với đường dẫn tác vụ đồng bộ hóa + truyền phát" +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -Quy trình làm việc của tổng đài viên cần cả phản hồi trực tiếp và thực thi theo luồng trong thời gian dài với khả năng kiểm soát vòng đời. +**How OmniRoute solves it:** -**Cách OmniRoute giải quyết vấn đề này:** +- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding -- Điểm cuối JSON-RPC A2A (`POST /a2a`) với `message/send` và `message/stream` -- Truyền phát SSE với sự lan truyền trạng thái đầu cuối -- API vòng đời tác vụ cho `tasks/get` và `tasks/cancel`
+ - -🛰️ 19. "Tôi cần tình trạng quy trình MCP thực sự, không phải trạng thái đoán" +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -Các nhóm vận hành cần biết liệu MCP có thực sự tồn tại hay không, chứ không chỉ là liệu API có thể truy cập được hay không. +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -**Cách OmniRoute giải quyết vấn đề này:** +**How OmniRoute solves it:** -- Tệp nhịp tim thời gian chạy với PID, dấu thời gian, vận chuyển, số lượng công cụ và chế độ phạm vi -- API trạng thái MCP kết hợp nhịp tim + hoạt động gần đây -- Thẻ trạng thái giao diện người dùng về độ mới của quy trình/thời gian hoạt động/nhịp tim
+- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` - -📋 20. "Tôi cần thực thi công cụ MCP có thể kiểm tra được" + -Khi các công cụ thay đổi cấu hình hoặc kích hoạt các hành động vận hành, các nhóm cần truy xuất nguồn gốc pháp lý. +
+🛰️ 19. "I need real MCP process health, not guessed status" -**Cách OmniRoute giải quyết vấn đề này:** +Operational teams need to know if MCP is actually alive, not just whether an API is reachable. -- Ghi nhật ký kiểm tra được hỗ trợ bởi SQLite cho các lệnh gọi công cụ MCP -- Bộ lọc theo công cụ, thành công/thất bại, khóa API và phân trang -- Bảng kiểm tra bảng điều khiển + điểm cuối thống kê để tự động hóa
+**How OmniRoute solves it:** - -🔐 21. "Tôi cần quyền MCP trong phạm vi mỗi lần tích hợp" +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -Các khách hàng khác nhau phải có quyền truy cập ít đặc quyền nhất vào các danh mục công cụ. + -**Cách OmniRoute giải quyết vấn đề này:** +
+📋 20. "I need auditable MCP tool execution" -- 10 phạm vi MCP chi tiết để truy cập công cụ được kiểm soát -- Thực thi phạm vi và khả năng hiển thị trong giao diện người dùng quản lý MCP -- Tư thế mặc định an toàn cho dụng cụ vận hành
+When tools mutate config or trigger ops actions, teams need forensic traceability. - -⚙️ 22. "Tôi cần kiểm soát hoạt động mà không cần triển khai lại" +**How OmniRoute solves it:** -Các nhóm cần thay đổi thời gian chạy nhanh trong các sự cố hoặc sự kiện tốn kém. +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -**Cách OmniRoute giải quyết vấn đề này:** + -- Chuyển đổi kích hoạt kết hợp trực tiếp từ bảng điều khiển MCP -- Áp dụng hồ sơ khả năng phục hồi từ các gói chính sách được xác định trước -- Đặt lại trạng thái ngắt mạch từ cùng bảng vận hành +
+🔐 21. "I need scoped MCP permissions per integration" - -🔄 23. "Tôi cần khả năng hiển thị và hủy trực tiếp trong vòng đời nhiệm vụ A2A" +Different clients should have least-privilege access to tool categories. -Nếu không có khả năng hiển thị vòng đời, các sự cố trong nhiệm vụ sẽ khó phân loại. +**How OmniRoute solves it:** -**Cách OmniRoute giải quyết vấn đề này:** +- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling -- Liệt kê/lọc nhiệm vụ theo trạng thái/kỹ năng với phân trang -- Xem chi tiết về siêu dữ liệu, sự kiện và hiện vật của nhiệm vụ -- Điểm cuối hủy tác vụ và hành động UI có xác nhận
+ - -🌊 24. "Tôi cần số liệu luồng hoạt động để tải A2A" +
+⚙️ 22. "I need operational controls without redeploying" -Luồng công việc phát trực tuyến yêu cầu hiểu biết sâu sắc về hoạt động đồng thời và kết nối trực tiếp. +Teams need quick runtime changes during incidents or cost events. -**Cách OmniRoute giải quyết vấn đề này:** +**How OmniRoute solves it:** -- Bộ đếm luồng hoạt động được tích hợp vào trạng thái A2A -- Dấu thời gian nhiệm vụ cuối cùng và số lượng trên mỗi trạng thái -- Thẻ bảng điều khiển A2A để theo dõi hoạt động theo thời gian thực
+- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel - -🪪 25. "Tôi cần khám phá đại lý tiêu chuẩn cho khách hàng" + -Máy khách và người điều phối bên ngoài cần siêu dữ liệu có thể đọc được bằng máy để triển khai. +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -**Cách OmniRoute giải quyết vấn đề này:** +Without lifecycle visibility, task incidents become hard to triage. -- Thẻ đại lý bị lộ tại `/.well-known/agent.json` -- Khả năng và kỹ năng thể hiện trong UI quản lý -- API trạng thái A2A bao gồm siêu dữ liệu khám phá để tự động hóa
+**How OmniRoute solves it:** - -🧭 26. "Tôi cần khả năng khám phá giao thức trong UX sản phẩm" +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -Nếu người dùng không thể khám phá các bề mặt giao thức, chất lượng chấp nhận và hỗ trợ sẽ giảm. + -**Cách OmniRoute giải quyết vấn đề này:** +
+🌊 24. "I need active stream metrics for A2A load" -- Trang**Điểm cuối**được hợp nhất với các tab dành cho Điểm cuối Proxy, MCP, A2A và API -- Chuyển đổi trạng thái dịch vụ nội tuyến (Trực tuyến/Ngoại tuyến) cho MCP và A2A -- Liên kết từ tổng quan đến các tab quản lý chuyên dụng
+Streaming workflows require operational insight into concurrency and live connections. - -🧪 27. "Tôi cần xác thực giao thức end-to-end với khách hàng thực" +**How OmniRoute solves it:** -Các thử nghiệm mô phỏng không đủ để xác thực tính tương thích của giao thức trước khi phát hành. +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -**Cách OmniRoute giải quyết vấn đề này:** + -- Bộ E2E khởi động ứng dụng và sử dụng vận chuyển máy khách MCP SDK thực -- Máy khách A2A kiểm tra các luồng khám phá, gửi, truyền phát, nhận và hủy -- Kiểm tra chéo các xác nhận đối với kiểm tra MCP và API nhiệm vụ A2A +
+🪪 25. "I need standard agent discovery for clients" - -📡 28. "Tôi cần khả năng quan sát thống nhất trên tất cả các giao diện" +External clients and orchestrators need machine-readable metadata for onboarding. -Việc phân chia khả năng quan sát theo giao thức sẽ tạo ra các điểm mù và MTTR dài hơn. +**How OmniRoute solves it:** -**Cách OmniRoute giải quyết vấn đề này:** +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation -- Bảng điều khiển/nhật ký/phân tích thống nhất trong một sản phẩm -- Sức khỏe + kiểm toán + yêu cầu đo từ xa trên các lớp OpenAI, MCP và A2A -- API hoạt động cho trạng thái và tự động hóa
+ - -💼 29. "Tôi cần một thời gian chạy cho proxy + công cụ + điều phối tác nhân" +
+🧭 26. "I need protocol discoverability in the product UX" -Việc chạy nhiều dịch vụ riêng biệt làm tăng chi phí vận hành và các chế độ lỗi. +If users cannot discover protocol surfaces, adoption and support quality drop. -**Cách OmniRoute giải quyết vấn đề này:** +**How OmniRoute solves it:** -- Proxy tương thích với OpenAI, máy chủ MCP và máy chủ A2A trong một ngăn xếp -- Chia sẻ xác thực, khả năng phục hồi, lưu trữ dữ liệu và khả năng quan sát -- Mô hình chính sách nhất quán trên tất cả các bề mặt tương tác
+- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs - -🚀 30. "Tôi cần gửi quy trình công việc tổng thể mà không cần sử dụng quá nhiều mã keo" + -Các nhóm bị mất tốc độ khi kết hợp nhiều dịch vụ và tập lệnh đặc biệt. +
+🧪 27. "I need end-to-end protocol validation with real clients" -**Cách OmniRoute giải quyết vấn đề này:** +Mock tests are not enough to validate protocol compatibility before release. -- Chiến lược điểm cuối thống nhất cho khách hàng và đại lý -- Giao diện người dùng quản lý giao thức tích hợp và đường dẫn xác thực khói -- Nền tảng sẵn sàng sản xuất (bảo mật, ghi nhật ký, khả năng phục hồi, sao lưu)
+**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + + + +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**Playbook A: Tối đa hóa đăng ký trả phí + dự phòng giá rẻ**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -608,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**Playbook B: Ngăn xếp mã hóa không tốn phí**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**Playbook C: chuỗi dự phòng luôn hoạt động 24/7**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -631,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**Playbook D: Tác nhân hoạt động với MCP + A2A**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> Thiết lập mã hóa AI trong vài phút với mức**$0/tháng**. Kết nối các tài khoản miễn phí này và sử dụng combo**Free Stack**tích hợp sẵn. +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -| Bước | Hành động | Nhà cung cấp đã được mở khóa | +| Step | Action | Providers Unlocked | | ---- | -------------------------------------------------- | ------------------------------------------------------------------ | -| 1 | Kết nối**Kiro**(ID AWS Builder OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**không giới hạn**| -| 2 | Kết nối**Qoder**(Google OAuth) | kimi-k2-thinking, qwen3-code-plus, deepseek-r1... —**không giới hạn**| -| 3 | Kết nối**Qwen**(Mã thiết bị) | qwen3-code-plus, qwen3-code-flash... —**không giới hạn**| -| 4 | Kết nối**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180K/tháng miễn phí**| -| 5 | `/dashboard/combos` →**Mẫu ngăn xếp miễn phí ($0)**| Tự động quay vòng tất cả các nhà cung cấp miễn phí | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**Trỏ bất kỳ IDE/CLI nào tới:**`http://localhost:20128/v1` · Khóa API: `any-string` · Xong. +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**Phạm vi phủ sóng bổ sung tùy chọn (cũng miễn phí):**Khóa API Groq (miễn phí 30 RPM), NVIDIA NIM (miễn phí 40 RPM, hơn 70 mẫu), Cerebras (1 triệu tok/ngày), khóa API LongCat (50 triệu mã thông báo/ngày!), Cloudflare Workers AI (10K nơ-ron/ngày, hơn 50 mô hình).## Bắt đầu nhanh +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## Bắt đầu nhanh ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **người dùng pnpm:**Chạy `pnpm confirm-builds -g` sau khi cài đặt để kích hoạt tập lệnh bản dựng gốc được yêu cầu bởi `better-sqlite3` và `@swc/core`: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > > ```bash -> cài đặt pnpm -g omniroute -> pnpm phê duyệt-builds -g # Chọn tất cả các gói → phê duyệt -> mọi tuyến đường +> pnpm install -g omniroute +> pnpm approve-builds -g # Select all packages → approve +> omniroute > ``` -Trang tổng quan mở tại `http://localhost:20128` và URL cơ sở API là `http://localhost:20128/v1`. +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| Lệnh | Mô tả | -| ----------------------- | --------------------------------------------------------------------------- | -| `toàn tuyến` | Máy chủ khởi động (`PORT=20128`, API và bảng điều khiển trên cùng một cổng) | -| `omniroute --port 3000` | Đặt cổng chuẩn/API thành 3000 | -| `omniroute --mcp` | Khởi động máy chủ MCP (stdio Transport) | -| `omniroute --no-open` | Không tự động mở trình duyệt | -| `toàn tuyến --help` | Hiển thị trợ giúp | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -Chế độ chia cổng tùy chọn:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -Đối với hầu hết các hoạt động triển khai, bạn chỉ cần: +For most deployments, you only need: -| Biến | Mặc định | Mục đích | -| ------------------------ | ----------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------ | -| `REQUEST_TIMEOUT_MS` | `600000` | Đường cơ sở được chia sẻ để tìm nạp ngược dòng, thời gian chờ Undici ẩn, yêu cầu vân tay TLS và yêu cầu cầu nối API/thời gian chờ proxy | -| `STREAM_IDLE_TIMEOUT_MS` | kế thừa `REQUEST_TIMEOUT_MS` | Khoảng cách tối đa giữa các đoạn phát trực tuyến trước khi OmniRoute hủy bỏ luồng SSE | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -Khả năng tương thích ngược được duy trì: `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` hiện có và các biến thời gian chờ trên mỗi lớp khác vẫn hoạt động và ghi đè đường cơ sở chung. +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -Ghi đè nâng cao có sẵn nếu bạn cần kiểm soát tốt hơn:| Biến | Mặc định | Mục đích | -| ---------------------------------------- | ------------------------------------------ | ----------------------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` | kế thừa `REQUEST_TIMEOUT_MS` | Tổng thời gian chờ yêu cầu ngược dòng được sử dụng bởi tín hiệu hủy bỏ tìm nạp chính | -| `FETCH_HEADERS_TIMEOUT_MS` | kế thừa `FETCH_TIMEOUT_MS` | Undici giới hạn thời gian nhận tiêu đề phản hồi ngược dòng | -| `FETCH_BODY_TIMEOUT_MS` | kế thừa `FETCH_TIMEOUT_MS` | Giới hạn thời gian Undici giữa các khối nội dung ngược dòng (`0` vô hiệu hóa nó) | -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Hết thời gian chờ kết nối TCP | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici hết thời gian chờ ổ cắm duy trì hoạt động | -| `TLS_CLIENT_TIMEOUT_MS` | kế thừa `FETCH_TIMEOUT_MS` | Hết thời gian chờ cho các yêu cầu vân tay TLS được thực hiện thông qua `wreq-js` | -| `API_BRIDGE_PROXY_TIMEOUT_MS` | kế thừa `REQUEST_TIMEOUT_MS` hoặc `30000` | Đã hết thời gian chờ chuyển tiếp proxy `/v1` từ cổng API sang cổng bảng điều khiển | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Hết thời gian chờ yêu cầu đến trên máy chủ cầu nối API | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Hết thời gian chờ tiêu đề đến trên máy chủ cầu nối API | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Thời gian chờ duy trì trên máy chủ cầu API | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Hết thời gian chờ không hoạt động của ổ cắm trên máy chủ cầu API (`0` tắt nó) | +Advanced overrides are available if you need finer control: -Nếu bạn chạy OmniRoute phía sau Nginx, Caddy, Cloudflare hoặc proxy ngược khác, hãy đảm bảo proxy -thời gian chờ cũng cao hơn thời gian chờ tìm nạp/phát trực tuyến OmniRoute của bạn.### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. Mở Bảng điều khiển → `Nhà cung cấp` và kết nối ít nhất một nhà cung cấp (khóa OAuth hoặc API). -2. Mở Bảng điều khiển → `Điểm cuối` và tạo khóa API. -3. (Tùy chọn) Mở Bảng điều khiển → `Combos` và đặt chuỗi dự phòng của bạn.### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -Hoạt động với Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode và SDK tương thích với OpenAI.### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP (đối với các hoạt động điều khiển bằng công cụ):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -Sau đó kết nối ứng dụng khách MCP của bạn qua `stdio` và các công cụ kiểm tra như: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A (dành cho quy trình làm việc giữa các đại lý):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -760,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -Bộ phần mềm này xác thực các luồng ứng dụng khách MCP và A2A thực dựa trên ứng dụng đang chạy.### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -768,14 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` - +
+Void Linux (`xbps-src` template) -Void Linux (mẫu `xbps-src`) - -Đối với người dùng Void Linux, bạn có thể xây dựng gói gốc bằng cách sử dụng `xbps-src`. Lưu khối này dưới dạng `srcpkgs/omniroute/template`:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -787,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -795,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -871,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -882,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute có sẵn dưới dạng hình ảnh Docker công khai trên [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**Chạy nhanh:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -892,85 +989,96 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**Với tệp môi trường:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` -```` +**Using Docker Compose:** -**Sử dụng Docker Compose:**```bash +```bash # Base profile (no CLI tools) docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -Hỗ trợ bảng điều khiển cho việc triển khai Docker hiện bao gồm**Đường hầm nhanh Cloudflare**chỉ bằng một cú nhấp chuột trên `Bảng điều khiển → Điểm cuối`. Đầu tiên, chỉ cho phép tải xuống `cloudflared` khi cần, bắt đầu một đường hầm tạm thời tới điểm cuối `/v1` hiện tại của bạn và hiển thị URL `https://*.trycloudflare.com/v1` được tạo ngay bên dưới URL công khai thông thường của bạn. +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -Ghi chú: +Notes: -- URL đường hầm nhanh là tạm thời và thay đổi sau mỗi lần khởi động lại. -- Đường hầm nhanh không được tự động khôi phục sau khi khởi động lại OmniRoute hoặc vùng chứa. Kích hoạt lại chúng từ bảng điều khiển khi cần. -- Cài đặt được quản lý hiện hỗ trợ Linux, macOS và Windows trên `x64` / `arm64`. -- Đường hầm nhanh được quản lý mặc định vận chuyển HTTP/2 để tránh cảnh báo bộ đệm QUIC UDP ồn ào trong môi trường vùng chứa bị hạn chế. Đặt `CLOUDFLARED_PROTOCOL=quic` hoặc `auto` nếu bạn muốn một phương tiện vận chuyển khác. -- Hình ảnh Docker gói các gốc CA của hệ thống và chuyển chúng tới `cloudflared` được quản lý, điều này tránh được lỗi tin cậy TLS khi đường hầm khởi động bên trong vùng chứa. -- SQLite chạy ở chế độ WAL. `docker stop` nên được phép kết thúc để OmniRoute có thể kiểm tra những thay đổi mới nhất trở lại `storage.sqlite`. -- Các tệp Compose đi kèm đã đặt thời gian gia hạn dừng là 40 giây. Nếu bạn chạy hình ảnh trực tiếp, hãy giữ `--stop-timeout 40` (hoặc tương tự) để việc dừng thủ công không cắt đứt quá trình dọn dẹp tắt máy. -- Đặt `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` nếu bạn muốn OmniRoute sử dụng tệp nhị phân hiện có thay vì tải xuống tệp nhị phân. +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**Sử dụng Docker Compose với Caddy (HTTPS Auto-TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -OmniRoute có thể được hiển thị một cách an toàn bằng cách sử dụng tính năng cung cấp SSL tự động của Caddy. Đảm bảo bản ghi DNS A của miền của bạn trỏ đến IP máy chủ của bạn.```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` - -| Hình ảnh | Gắn thẻ | Kích thước | Mô tả | +| Image | Tag | Size | Description | | ------------------------ | -------- | ------ | --------------------- | -| `diegosouzapw/omniroute` | `mới nhất` | ~250MB | Bản phát hành ổn định mới nhất | -| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Phiên bản hiện tại |--- +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | + +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**MỚI!**OmniRoute hiện có sẵn dưới dạng**ứng dụng máy tính để bàn gốc**dành cho Windows, macOS và Linux. +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -Chạy OmniRoute dưới dạng một ứng dụng máy tính để bàn độc lập — không cần thiết bị đầu cuối, không cần trình duyệt, không cần Internet đối với các mô hình cục bộ. Ứng dụng dựa trên Electron bao gồm: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**Cửa sổ gốc**— Cửa sổ ứng dụng chuyên dụng có tích hợp khay hệ thống -- 🔄**Tự động khởi động**— Khởi chạy OmniRoute khi đăng nhập hệ thống -- 🔔**Thông báo gốc**— Nhận thông báo về tình trạng cạn kiệt hạn ngạch hoặc các vấn đề về nhà cung cấp -- ⚡**Cài đặt bằng một cú nhấp chuột**— NSIS (Windows), DMG (macOS), AppImage (Linux) -- 🌐**Chế độ ngoại tuyến**— Hoạt động hoàn toàn ngoại tuyến với máy chủ đi kèm### Bắt đầu nhanh +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### Bắt đầu nhanh ```bash # Development mode @@ -981,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -Khi được thu nhỏ, OmniRoute sẽ tồn tại trong khay hệ thống của bạn bằng các hành động nhanh chóng: +When minimized, OmniRoute lives in your system tray with quick actions: -- Mở bảng điều khiển -- Thay đổi cổng máy chủ -- Thoát khỏi ứng dụng +- Open dashboard +- Change server port +- Quit application -📖 Tài liệu đầy đủ: [`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -| Bậc | Nhà cung cấp | Chi phí | Đặt lại hạn ngạch | Tốt nhất cho | -| --------------- | --------------------------- | -------------------------------- | ---------------------- | ----------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **💳 ĐĂNG KÝ** | Mã Claude (Pro) | $20/tháng | 5h + hàng tuần | Đã đăng ký | -| | Codex (Plus/Pro) | $20-200/tháng | 5h + hàng tuần | Người dùng OpenAI | -| | Song Tử CLI | **MIỄN PHÍ** | 180K/tháng + 1K/ngày | Mọi người! | -| | Phi công phụ GitHub | $10-19/tháng | Hàng tháng | Người dùng GitHub | -| **🔑 KHÓA API** | NVIDIA NIM | **MIỄN PHÍ**(dev mãi mãi) | ~40 vòng/phút | Hơn 70 mô hình mở | -| | Não | **MIỄN PHÍ**(1 triệu tok/ngày) | 60K TPM / 30 vòng/phút | Nhanh nhất thế giới | -| | Groq | **MIỄN PHÍ**(30 vòng/phút) | 14,4K RPD | Llama/Gemma cực nhanh | -| | DeepSeek V3.2 | 0,27 USD/1,10 USD mỗi 1 triệu | Không có | Lý luận về giá/chất lượng tốt nhất | -| | xAI Grok-4 Nhanh | **$0,20/$0,50 mỗi 1 triệu**🆕 | Không có | Gọi công cụ + nhanh nhất, cực nhanh | -| | xAI Grok-4 (tiêu chuẩn) | 0,20 USD/1,50 USD mỗi 1 triệu 🆕 | Không có | Lý luận hàng đầu từ xAI | -| | Mistral | Dùng thử miễn phí + trả phí | Tỷ lệ giới hạn | AI Châu Âu | -| | OpenRouter | Trả tiền cho mỗi lần sử dụng | Không có | Tổng hợp hơn 100 mô hình | -| **💰 RẺ** | GLM-5 (thông qua Z.AI) 🆕 | 0,5 USD/1 triệu USD | 10 giờ sáng hàng ngày | Đầu ra 128K, chiếc hạm mới nhất | -| | GLM-4.7 | 0,6 USD/1 triệu USD | 10 giờ sáng hàng ngày | Dự phòng ngân sách | -| | MiniMax M2.5 🆕 | 0,3 USD/đầu vào 1 triệu USD | lăn 5 giờ | Lý luận + nhiệm vụ tác nhân | -| | MiniMax M2.1 | 0,2 USD/1 triệu USD | lăn 5 giờ | Lựa chọn rẻ nhất | -| | Kimi K2.5 (API Moonshot) 🆕 | Trả tiền cho mỗi lần sử dụng | Không có | Truy cập API Moonshot trực tiếp | -| | Kimi K2 | $9/tháng căn hộ | 10 triệu token/tháng | Chi phí dự đoán | -| **🆓 MIỄN PHÍ** | Qoder | **$0** | Không giới hạn | 5 mẫu không giới hạn | -| | Qwen | **$0** | Không giới hạn | 4 mẫu không giới hạn | -| | Kiro | **$0** | Không giới hạn | Claude Sonnet/Haiku (Người xây dựng AWS) | -| | LongCat Flash-Lite 🆕 | **$0**(50 triệu tok/ngày 🔥) | 1 RPS | Hạn ngạch miễn phí lớn nhất trên Trái đất | -| | Thụ phấn AI 🆕 | **$0**(không cần chìa khóa) | 1 yêu cầu/15 giây | GPT-5, Claude, DeepSeek, Llama 4 | -| | Cloudflare Workers AI 🆕 | **$0**(10K nơ-ron/ngày) | ~150 lần/ngày | Hơn 50 mẫu, lợi thế toàn cầu | -| | Đường quy mô AI 🆕 | **$0**(Tổng số 1 triệu token) | Tỷ lệ giới hạn | EU/GDPR, Qwen3 235B, Llama 70B | > 🆕**Thêm các mẫu mới (tháng 3 năm 2026):**Dòng Grok-4 Fast ở mức 0,20 USD/0,50 USD/M (điểm chuẩn ở 1143 mili giây — nhanh hơn 30% so với Gemini 2.5 Flash), GLM-5 qua Z.AI với đầu ra 128K, lý luận MiniMax M2.5, giá cập nhật DeepSeek V3.2, Kimi K2.5 qua API trực tiếp Moonshot. | +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 Ngăn xếp combo $0 — Thiết lập miễn phí hoàn chỉnh:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**Không tốn phí. Không bao giờ ngừng mã hóa.**Định cấu hình tính năng này dưới dạng một tổ hợp OmniRoute và tất cả các dự phòng sẽ tự động diễn ra — không cần chuyển đổi thủ công.--- +--- --- ## 🆓 Free Models — What You Actually Get -> Tất cả các mẫu bên dưới đều**miễn phí 100% và không cần thẻ tín dụng**. OmniRoute tự động định tuyến giữa chúng khi hết một hạn mức — kết hợp tất cả chúng để tạo thành một combo $0 không thể phá vỡ.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -| Người mẫu | Tiền tố | Giới hạn | Giới hạn tỷ lệ | +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | --------------------- | -| `claude-sonnet-4.5` | `kr/` |**Không giới hạn**| Không có giới hạn hàng ngày được báo cáo | -| `claude-haiku-4.5` | `kr/` |**Không giới hạn**| Không có giới hạn hàng ngày được báo cáo | -| `claude-opus-4.6` | `kr/` |**Không giới hạn**| Opus mới nhất qua Kiro |### 🟢 QODER MODELS (Free PAT via qodercli) +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -| Người mẫu | Tiền tố | Giới hạn | Giới hạn tỷ lệ | +### 🟢 QODER MODELS (Free PAT via qodercli) + +| Model | Prefix | Limit | Rate Limit | | ------------------ | ------ | ------------- | --------------- | -| `kimi-k2-suy nghĩ` | `nếu/` |**Không giới hạn**| Không có giới hạn được báo cáo | -| `qwen3-code-plus` | `nếu/` |**Không giới hạn**| Không có giới hạn được báo cáo | -| `deepseek-r1` | `nếu/` |**Không giới hạn**| Không có giới hạn được báo cáo | -| `minimax-m2.1` | `nếu/` |**Không giới hạn**| Không có giới hạn được báo cáo | -| `kimi-k2` | `nếu/` |**Không giới hạn**| Không có giới hạn được báo cáo | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | -> Phương thức kết nối được đề xuất:**Mã thông báo truy cập cá nhân + `qodercli`**. OAuth của trình duyệt là -> thử nghiệm và bị tắt theo mặc định trừ khi các biến môi trường `QODER_OAUTH_*` được định cấu hình.### 🟡 QWEN MODELS (Device Code Auth) +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. -| Người mẫu | Tiền tố | Giới hạn | Giới hạn tỷ lệ | +### 🟡 QWEN MODELS (Device Code Auth) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | ------------------- | -| `qwen3-code-plus` | `qw/` |**Không giới hạn**| Không có giới hạn được báo cáo | -| `qwen3-code-flash` | `qw/` |**Không giới hạn**| Không có giới hạn được báo cáo | -| `qwen3-code-next` | `qw/` |**Không giới hạn**| Không có giới hạn được báo cáo | -| `mô hình tầm nhìn` | `qw/` |**Không giới hạn**| Đa phương thức (hình ảnh) |### 🟣 GEMINI CLI (Google OAuth) +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -| Người mẫu | Tiền tố | Giới hạn | Giới hạn tỷ lệ | -| ------------------------ | ------ | ----------------------------- | ------------- | -| `gemini-3-flash-xem trước` | `gc/` |**180K tok/tháng**+ 1K/ngày | Đặt lại hàng tháng | -| `gemini-2.5-pro` | `gc/` | 180K/tháng (nhóm chung) | Chất lượng cao |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +### 🟣 GEMINI CLI (Google OAuth) -| Bậc | Giới hạn hàng ngày | Giới hạn tỷ lệ | Ghi chú | +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | + +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---------- | ------------ | ----------- | ------------------------------------------------------ | -| Miễn phí (Nhà phát triển) | Không có giới hạn mã thông báo |**~40 vòng/phút**| Hơn 70 mẫu; chuyển sang giới hạn lãi suất thuần túy vào giữa năm 2025 | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -Các mẫu miễn phí phổ biến: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -| Bậc | Giới hạn hàng ngày | Giới hạn tỷ lệ | Ghi chú | -| ---- | ----------------- | ---------------- | --------------------------------------------- | -| Miễn phí |**1 triệu token/ngày**| 60K TPM / 30 vòng/phút | Suy luận LLM nhanh nhất thế giới; đặt lại hàng ngày | +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) -Có sẵn miễn phí: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ----------------- | ---------------- | ------------------------------------------- | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -| Bậc | Giới hạn hàng ngày | Giới hạn tỷ lệ | Ghi chú | +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` + +### 🔴 GROQ (Free API Key — console.groq.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---- | ------------- | ---------------- | ----------------------------------------- | -| Miễn phí |**14,4K RPD**| 30 vòng/phút cho mỗi mẫu | Không có thẻ tín dụng; Giới hạn 429, không bị tính phí | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -Có sẵn miễn phí: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -| Người mẫu | Tiền tố | Hạn ngạch miễn phí hàng ngày | Ghi chú | -| ----------------------------- | ------ | ----------------- | -------------- | -| `LongCat-Flash-Lite` | `lc/` |**50 triệu token**🔥 | Hạn ngạch miễn phí lớn nhất từ ​​trước đến nay | -| `LongCat-Flash-Chat` | `lc/` | 500K token | Trò chuyện nhiều lượt | -| `LongCat-Tư duy chớp nhoáng` | `lc/` | 500K token | Lý luận / CoT | -| `LongCat-Flash-Tư duy-2601` | `lc/` | 500K token | Phiên bản tháng 1 năm 2026 | -| `LongCat-Flash-Omni-2603` | `lc/` | 500K token | Đa phương thức | +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 -> Miễn phí 100% khi ở phiên bản beta công khai. Đăng ký tại [longcat.chat](https://longcat.chat) bằng email hoặc điện thoại. Đặt lại 00:00 UTC hàng ngày.### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | -| Người mẫu | Tiền tố | Giới hạn tỷ lệ | Nhà cung cấp đằng sau | +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. + +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| `openai` | `pol/` | 1 yêu cầu/15 giây | GPT-5 | -| `claude` | `pol/` | 1 yêu cầu/15 giây | Claude nhân loại | -| `song tử` | `pol/` | 1 yêu cầu/15 giây | Google Song Tử | -| `tìm kiếm sâu` | `pol/` | 1 yêu cầu/15 giây | DeepSeek V3 | -| `llama` | `pol/` | 1 yêu cầu/15 giây | Hướng đạo Meta Llama 4 | -| `mistral` | `pol/` | 1 yêu cầu/15 giây | AI của Mistral | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**Không ma sát:**Không cần đăng ký, không cần khóa API. Thêm nhà cung cấp Pollinations với trường khóa trống và nó sẽ hoạt động ngay lập tức.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -| Bậc | Tế bào thần kinh hàng ngày | Cách sử dụng tương đương | Ghi chú | -| ---- | ------------- | ------------------------------ | -------------- | -| Miễn phí |**10.000**| ~150 LLM resp / âm thanh 500 giây / 15K lượt nhúng | Cạnh toàn cầu, hơn 50 mẫu | +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 -Các mô hình miễn phí phổ biến: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (âm thanh miễn phí!), `@cf/qwen/qwen2.5-code-15b-instruct` +| Tier | Daily Neurons | Equivalent Usage | Notes | +| ---- | ------------- | --------------------------------------- | ----------------------- | +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -> Yêu cầu Mã thông báo API + ID tài khoản từ [dash.cloudflare.com](https://dash.cloudflare.com). Lưu trữ ID tài khoản trong cài đặt nhà cung cấp.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -| Bậc | Hạn ngạch miễn phí | Vị trí | Ghi chú | +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. + +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | | ---- | ------------- | ------------ | ----------------------------------- | -| Miễn phí |**1 triệu token**| 🇫🇷 Paris, EU | Không cần thẻ tín dụng trong giới hạn | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | -Có sẵn miễn phí: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` -> Tuân thủ EU/GDPR. Nhận khóa API tại [console.scaleway.com](https://console.scaleway.com). +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). ->**💡 Kho miễn phí tối ưu (11 nhà cung cấp, $0 vĩnh viễn):** +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -> Qoder (if/) → kimi-k2-thinking, qwen3-code-plus, deepseek-r1 UNLIMITED -> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 triệu token/ngày 🔥 -> Thụ phấn (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — không cần chìa khóa -> Qwen (qw/) → mô hình qwen3-code UNLIMITED -> Gemini (gemini/) → Gemini 2.5 Flash — miễn phí 1.500 yêu cầu/ngày -> Cloudflare AI (cf/) → Hơn 50 mô hình — 10K nơ-ron/ngày -> Đường mở rộng (scw/) → Qwen3 235B, Llama 70B — 1 triệu token miễn phí (EU) -> Groq (groq/) → Llama/Gemma — 14,4K yêu cầu/ngày cực nhanh -> NVIDIA NIM (nvidia/) → hơn 70 mẫu mở — 40 RPM mãi mãi -> Cerebras (cerebras/) → Llama/Qwen nhanh nhất thế giới — 1 triệu tok/ngày -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> Phiên âm mọi âm thanh/video với giá**$0**— Deepgram dẫn đầu với 200 USD miễn phí, dự phòng AssemblyAI 50 USD, Groq Whisper làm bản sao lưu khẩn cấp không giới hạn. +## 🎙️ Free Transcription Combo -| Nhà cung cấp | Tín dụng miễn phí | Người Mẫu Tốt Nhất | Giới hạn tỷ lệ | +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. + +| Provider | Free Credits | Best Model | Rate Limit | | ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | -| 🟢**Deepgram**|**$200 miễn phí**(đăng ký) | `nova-3` — độ chính xác tốt nhất, hơn 30 ngôn ngữ | Không có giới hạn RPM đối với tín dụng miễn phí | -| 🔵**HộiAI**|**$50 miễn phí**(đăng ký) | `universal-3-pro` — chương, tình cảm, PII | Không có giới hạn RPM đối với tín dụng miễn phí | -| 🔴**Ngốc nghếch**|**Miễn phí mãi mãi**| `thì thầm-large-v3` — OpenAI Whisper | 30 vòng/phút (tốc độ giới hạn) | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | -**Kết hợp được đề xuất trong `/dashboard/combos`:**``` +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -Sau đó, trong tab `/dashboard/media` →**Phiên âm**: tải lên bất kỳ tệp âm thanh hoặc video nào → chọn điểm cuối kết hợp của bạn → nhận phiên âm ở các định dạng được hỗ trợ.## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 được xây dựng như một nền tảng hoạt động, không chỉ là proxy chuyển tiếp.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| Tính năng | Nó làm gì | -| ----------------------------------------- | --------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Gia đình nhanh Grok-4** | mô hình xAI ở mức 0,20 USD/0,50 USD/M — tốc độ chuẩn là 1143 mili giây (nhanh hơn 30% so với Gemini 2.5 Flash) | -| 🧠**GLM-5 qua Z.AI** | Bối cảnh đầu ra 128K, 0,5 USD/1 triệu USD — sản phẩm chủ lực mới nhất của dòng GLM | -| 🔮**MiniMax M2.5** | Lý luận + nhiệm vụ tác nhân ở mức 0,30 USD/1 triệu — nâng cấp đáng kể từ M2.1 | -| 🎯**công cụCờ gọi theo mẫu** | Trên mỗi mô hình `toolCalling: true/false` trong sổ đăng ký — AutoCombo bỏ qua các mô hình không hỗ trợ công cụ | -| 🌍**Phát hiện ý định đa ngôn ngữ** | Từ khóa PT/ZH/ES/AR trong tính điểm AutoCombo — lựa chọn mô hình tốt hơn cho nội dung không phải tiếng Anh | -| 📊**Dự phòng dựa trên điểm chuẩn** | Độ trễ p95 thực từ tính điểm kết hợp nguồn cấp dữ liệu yêu cầu trực tiếp — AutoCombo học hỏi từ dữ liệu thực tế | -| 🔁**Yêu cầu loại bỏ trùng lặp** | Cửa sổ khấu trừ dựa trên hàm băm nội dung — an toàn cho nhiều tác nhân, ngăn chặn các khoản phí trùng lặp | -| 🔌**Chiến lược bộ định tuyến có thể cắm** | Giao diện `RouterStrategy` có thể mở rộng - thêm logic định tuyến tùy chỉnh làm plugin | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| Tính năng | Nó làm gì | -| ---------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**Sân chơi kiểu mẫu** | Trang tổng quan để kiểm tra trực tiếp bất kỳ mô hình nào — bộ chọn nhà cung cấp/mô hình/điểm cuối, Trình chỉnh sửa Monaco, phát trực tuyến, hủy bỏ, tính thời gian | -| 🔏**Khớp vân tay CLI** | Thứ tự tiêu đề/nội dung của mỗi nhà cung cấp để khớp với chữ ký CLI gốc — chuyển đổi cho mỗi nhà cung cấp trong Cài đặt > Bảo mật.**IP proxy của bạn được giữ nguyên** | -| 🤝**Hỗ trợ ACP (Giao thức khách hàng đại lý)** | Khám phá tác nhân CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 cái khác), trình tạo quy trình, điểm cuối `/api/acp/agents` | -| 🤖**Bảng thông tin đại lý ACP** | Trang Gỡ lỗi > Tác nhân — lưới gồm 14 tác nhân với trạng thái cài đặt, phiên bản, biểu mẫu tác nhân tùy chỉnh cho bất kỳ công cụ CLI nào.**Người dùng OpenCode**nhận được nút "Tải xuống opencode.json" để tự động tạo cấu hình sẵn sàng sử dụng với tất cả các mẫu có sẵn. | -| 🔧**Định tuyến mô hình tùy chỉnh `apiFormat`** | Các mô hình tùy chỉnh với `apiFormat: "responses"` hiện định tuyến chính xác đến trình dịch API Phản hồi | -| 🏢**Cách ly không gian làm việc Codex** | Nhiều không gian làm việc Codex cho mỗi email - OAuth phân tách chính xác các kết nối theo ID không gian làm việc | -| 🔄**Tự động cập nhật điện tử** | Ứng dụng máy tính để bàn kiểm tra các bản cập nhật + tự động cài đặt khi khởi động lại | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| Tính năng | Nó làm gì | -| ---------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**Máy chủ MCP (25 công cụ)** | Công cụ IDE/agent thông qua 3 phương thức truyền tải: stdio, SSE (`/api/mcp/sse`), HTTP có thể phát trực tuyến (`/api/mcp/stream`). 18 lõi + 3 bộ nhớ + 4 công cụ kỹ năng | -| 🤝**Máy chủ A2A (JSON-RPC + SSE)** | Thực thi nhiệm vụ giữa các tác nhân với các luồng đồng bộ hóa và truyền phát | -| 🧭**Trang điểm cuối tổng hợp** | Trang quản lý theo thẻ với các tab Endpoint Proxy, MCP, A2A và API Endpoints | -| 🎚️**Bật/Tắt dịch vụ** | Công tắc BẬT/TẮT cho MCP và A2A với khả năng duy trì cài đặt (mặc định: TẮT) | -| 🛰️**Nhịp tim thời gian chạy MCP** | Trạng thái quy trình thực (pid, thời gian hoạt động, tuổi nhịp tim, vận chuyển, chế độ phạm vi) | -| 📋**Dấu vết kiểm tra MCP** | Nhật ký kiểm tra có thể lọc với thành công/thất bại và phân bổ chính | -| 🔐**Thực thi phạm vi MCP** | 10 quyền phạm vi chi tiết để truy cập công cụ được kiểm soát | -| 📡**Quản lý vòng đời nhiệm vụ A2A** | Liệt kê/lọc nhiệm vụ, kiểm tra sự kiện/hiện vật, hủy nhiệm vụ đang chạy | -| 📋**Agent Card Discovery** | `/.well-known/agent.json` để tự động phát hiện ứng dụng khách | -| 🧪**Khai thác thử nghiệm giao thức E2E** | Luồng máy khách MCP SDK + A2A thực trong `test:protocols:e2e` | -| ⚙️**Kiểm soát hoạt động** | Chuyển đổi tổ hợp, áp dụng cấu hình khả năng phục hồi, đặt lại bộ ngắt từ một bề mặt điều khiển | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| Tính năng | Nó làm gì | -| ---------------------------------------------- | ----------------------------------------------------------------------------------- | ----------------------- | -| 🎯**Dự phòng 4 tầng thông minh** | Tự động định tuyến: Đăng ký → Khóa API → Giá rẻ → Miễn phí | -| 📊**Theo dõi hạn ngạch theo thời gian thực** | Số lượng mã thông báo trực tiếp + đếm ngược đặt lại cho mỗi nhà cung cấp | -| 🔄**Dịch định dạng** | OpenAI ↔ Claude ↔ Gemini ↔ Phản hồi bằng chuyển đổi an toàn lược đồ | -| 👥**Hỗ trợ nhiều tài khoản** | Nhiều tài khoản cho mỗi nhà cung cấp với lựa chọn thông minh | -| 🔄**Tự động làm mới mã thông báo** | Mã thông báo OAuth tự động làm mới bằng thử lại | -| 🎨**Combo tùy chỉnh** | 9 chiến lược cân bằng + kiểm soát chuỗi dự phòng | -| 🌐**Bộ định tuyến ký tự đại diện** | `nhà cung cấp/*` định tuyến động | -| 🧠**Suy nghĩ về việc kiểm soát ngân sách** | Giới hạn lý luận truyền qua, tự động, tùy chỉnh và thích ứng | -| 🔀**Bí danh người mẫu** | Tích hợp sẵn + đặt bí danh mô hình tùy chỉnh và an toàn di chuyển | -| ⚡**Xuống cấp nền** | Định tuyến các tác vụ nền có mức độ ưu tiên thấp tới các mô hình rẻ hơn | -| 🧪**Định tuyến thông minh nhận biết nhiệm vụ** | Tự động chọn mô hình theo loại nội dung (mã hóa/tầm nhìn/phân tích/tóm tắt) | -| 🔄**Quy trình làm việc của đại lý A2A** | Bộ điều phối FSM xác định để thực thi tác nhân nhiều bước có trạng thái | -| 🔀**Định tuyến thích ứng** | Ghi đè chiến lược động dựa trên khối lượng mã thông báo và độ phức tạp của lời nhắc | -| 🎲**Đa dạng nhà cung cấp** | Shannon tính điểm entropy cân bằng phân phối lưu lượng truy cập tự động kết hợp | -| 💬**Tiêm nhắc nhở hệ thống** | Kiểm soát hành vi toàn cầu được áp dụng nhất quán | -| 📄**Khả năng tương thích API phản hồi** | Hỗ trợ đầy đủ `/v1/responses` cho Codex và quy trình làm việc tác nhân nâng cao | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| Tính năng | Nó làm gì | -| ------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**Tạo hình ảnh** | `/v1/images/thế hệ` với phần phụ trợ cục bộ và đám mây | -| 📐**Nhúng** | `/v1/embeddings` cho đường ống tìm kiếm và RAG | -| 🎤**Phiên âm âm thanh** | `/v1/audio/transcriptions` — 7 nhà cung cấp (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), phát hiện ngôn ngữ tự động, hỗ trợ MP4/MP3/WAV | -| 🔊**Chuyển văn bản thành giọng nói** | `/v1/audio/speech` — 10 nhà cung cấp (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) với thông báo lỗi chính xác | -| 🎬**Tạo video** | `/v1/video/thế hệ` (Quy trình làm việc ComfyUI + SD WebUI) | -| 🎵**Thế hệ âm nhạc** | `/v1/music/thế hệ` (Quy trình làm việc của ComfyUI) | -| 🛡️**Kiểm duyệt** | kiểm tra an toàn `/v1/modations` | -| 🔀**Sắp xếp lại** | `/v1/rerank` để chấm điểm mức độ phù hợp | -| 🔍**Tìm kiếm trên web**🆕 | `/v1/search` — 5 nhà cung cấp (Serper, Brave, Perplexity, Exa, Tavily), hơn 6.500 miễn phí/tháng, tự động chuyển đổi dự phòng, bộ đệm | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| Tính năng | Nó làm gì | -| ------------------------------------------ | ------------------------------------------------------------------------------------------------------------------ | -------------------------------- | -| 🔌**Bộ ngắt mạch** | Chuyến đi/khôi phục theo từng mô hình với các điều khiển ngưỡng | -| 🎯**Mô hình nhận biết điểm cuối** | Mô hình tùy chỉnh khai báo điểm cuối được hỗ trợ + định dạng API | -| 🛡️**Bầy chống sấm sét** | Bảo vệ Mutex + semaphore đối với các sự kiện thử lại/đánh giá | -| 🧠**Bộ đệm ngữ nghĩa + chữ ký** | Giảm chi phí/độ trễ với hai lớp bộ đệm | -| ⚡**Yêu cầu quyền bình đẳng** | Cửa sổ bảo vệ trùng lặp | -| 🔒**Giả mạo vân tay TLS** | Dấu vân tay TLS giống trình duyệt —**giảm khả năng phát hiện bot và gắn cờ tài khoản** | -| 🔏**Khớp vân tay CLI** | Phù hợp với chữ ký yêu cầu CLI gốc —**giảm rủi ro bị cấm trong khi vẫn bảo toàn IP proxy** | -| 🌐**Lọc IP** | Kiểm soát danh sách cho phép/danh sách chặn đối với các triển khai được hiển thị | -| 📊**Giới hạn tỷ lệ có thể chỉnh sửa** | Giới hạn cấp độ nhà cung cấp/toàn cầu có thể định cấu hình với tính bền bỉ | -| 📉**Sự xuống cấp duyên dáng** | Dự phòng khả năng nhiều lớp bảo vệ hoạt động của cổng cốt lõi | -| 📜**Đường kiểm tra cấu hình** | Theo dõi thay đổi dựa trên sự khác biệt ngăn chặn tình trạng trôi dạt trong hoạt động bằng cách khôi phục đơn giản | -| ⏳**Đồng bộ hóa sức khỏe nhà cung cấp** | Giám sát hết hạn mã thông báo chủ động kích hoạt cảnh báo trước khi xảy ra lỗi ủy quyền | -| 🚪**Tự động vô hiệu hóa tài khoản bị cấm** | Bộ ngắt mạch hoạt động tự động niêm phong các tài khoản token bị chặn vĩnh viễn | -| 🔑**Quản lý khóa API + Phạm vi** | Bảo mật việc phát hành/xoay vòng khóa và kiểm soát mô hình/nhà cung cấp | -| 👁️**Tiết lộ khóa API có phạm vi**🆕 | Chọn tham gia khôi phục khóa API thông qua `ALLOW_API_KEY_REVEAL` | -| 🛡️**Được bảo vệ `/model`** | Tùy chọn xác thực và ẩn nhà cung cấp cho danh mục mô hình | ### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| Tính năng | Nó làm gì | -| --------------------------------------------- | ------------------------------------------------------------------------------ | ---------------------------- | -| 📝**Yêu cầu + Ghi nhật ký proxy** | Yêu cầu/phản hồi đầy đủ và ghi nhật ký proxy | -| 📉**Nhật ký chi tiết được phát trực tuyến**🆕 | Tái cấu trúc các luồng tải trọng SSE một cách rõ ràng vào giao diện người dùng | -| 📋**Bảng điều khiển nhật ký hợp nhất** | Chế độ xem yêu cầu, proxy, kiểm tra và bảng điều khiển trong một trang | -| 🔍**Yêu cầu đo từ xa** | độ trễ p50/p95/p99 và theo dõi yêu cầu | -| 🏥**Bảng thông tin sức khỏe** | Thời gian hoạt động, trạng thái ngắt, khóa, số liệu thống kê bộ đệm | -| 💰**Theo dõi chi phí** | Kiểm soát ngân sách và khả năng hiển thị giá theo từng mô hình | -| 📈**Trực quan hóa phân tích** | Thông tin chi tiết về cách sử dụng mô hình/nhà cung cấp và lượt xem xu hướng | -| 🧪**Khung đánh giá** | Thử nghiệm bộ vàng với các chiến lược kết hợp có thể định cấu hình | -| 📡**Chẩn đoán trực tiếp**🆕 | Bỏ qua bộ nhớ đệm ngữ nghĩa để kiểm tra trực tiếp kết hợp chính xác | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| Tính năng | Nó làm gì | -| ---------------------------------- | ------------------------------------------------------------------------------------ | --------------------- | -| 🌐**Triển khai mọi nơi** | Localhost, VPS, Docker, Môi trường đám mây | -| 🚇**Đường hầm Cloudflare**🆕 | Tích hợp Đường hầm nhanh chỉ bằng một cú nhấp chuột từ bảng điều khiển | -| 🔑**Lọc mô hình khóa API** | Phản hồi gốc /v1/models được lọc thông qua các vai trò ngữ cảnh Bearer được chỉ định | -| ⚡**Bỏ qua bộ nhớ đệm thông minh** | Chẩn đoán TTL có thể định cấu hình và điều khiển tìm nạp lại bắt buộc | -| 🔄**Sao lưu/Khôi phục** | Luồng xuất/nhập khẩu và khắc phục thảm họa | -| 🧙**Trình hướng dẫn giới thiệu** | Thiết lập có hướng dẫn lần đầu | -| 🔧**Bảng điều khiển công cụ CLI** | Thiết lập bằng một cú nhấp chuột cho các công cụ mã hóa phổ biến | -| 🎮**Sân chơi kiểu mẫu** | Kiểm tra bất kỳ nhà cung cấp/mô hình/điểm cuối nào từ bảng điều khiển | -| 🔏**Chuyển đổi vân tay CLI** | So khớp dấu vân tay của mỗi nhà cung cấp trong Cài đặt > Bảo mật | -| 🌐**i18n (30 ngôn ngữ)** | Hỗ trợ ngôn ngữ tài liệu + bảng điều khiển đầy đủ với phạm vi bảo hiểm RTL | -| 🧹**Xóa tất cả các mẫu** | Xóa danh sách mô hình bằng một cú nhấp chuột trong chi tiết nhà cung cấp | -| 👁️**Điều khiển thanh bên**🆕 | Ẩn các thành phần và tích hợp khỏi Cài đặt giao diện | -| 📋**Mẫu vấn đề** | Mẫu GitHub được chuẩn hóa cho các lỗi và tính năng | -| 📂**Thư mục dữ liệu tùy chỉnh** | ghi đè `DATA_DIR` cho vị trí lưu trữ | ### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1294,103 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -Khi hạn ngạch, tỷ lệ hoặc tình trạng không thành công, OmniRoute sẽ tự động chuyển sang ứng viên tiếp theo mà không cần chuyển đổi thủ công.#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A có thể được tìm thấy trong giao diện người dùng và tài liệu (không bị ẩn) -- API trạng thái giao thức hiển thị dữ liệu hoạt động trực tiếp (`/api/mcp/*`, `/api/a2a/*`) -- Bảng điều khiển bao gồm các hành động cho hoạt động ngày thứ 2 (chuyển đổi kết hợp, đặt lại bộ ngắt, hủy tác vụ)#### Translator + validation workflow +#### Protocol management that is visible and operable -Khu vực dịch bao gồm: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Sân chơi**: yêu cầu kiểm tra chuyển đổi -**Bộ kiểm tra trò chuyện**: toàn bộ yêu cầu/phản hồi khứ hồi -**Băng thử nghiệm**: nhiều trường hợp trong một lần chạy -**Giám sát trực tiếp**: chế độ xem giao thông theo thời gian thực +#### Translator + validation workflow -Cộng thêm xác thực giao thức với các máy khách thực thông qua `npm run test:protocols:e2e`. +The Translator area includes: -> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Tham chiếu công cụ, cấu hình IDE và ví dụ ứng dụng khách +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A Server README](src/lib/a2a/README.md)**— Kỹ năng, phương pháp JSON-RPC, phát trực tuyến và vòng đời tác vụ## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute bao gồm khung đánh giá tích hợp để kiểm tra chất lượng phản hồi LLM dựa trên bộ vàng. Truy cập thông qua**Analytics → Đánh giá**trong bảng điều khiển.### Built-in Golden Set +## 🧪 Evaluations (Evals) -"Bộ vàng OmniRoute" được tải sẵn chứa các trường hợp thử nghiệm cho: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- Lời chào, toán, địa lý, tạo mã -- Tuân thủ định dạng JSON, dịch thuật, tạo đánh dấu -- Từ chối an toàn (nội dung có hại), đếm, logic boolean### Evaluation Strategies +### Built-in Golden Set -| Chiến lược | Mô tả | Ví dụ | -| ----------- | --------------------------------------------------------------- | -------------------------------- | --- | -| `chính xác` | Đầu ra phải khớp chính xác | `"4"` | -| `chứa` | Đầu ra phải chứa chuỗi con (không phân biệt chữ hoa chữ thường) | `"Paris"` | -| `regex` | Đầu ra phải khớp với mẫu biểu thức chính quy | `"1.*2.*3"` | -| `tùy chỉnh` | Hàm JS tùy chỉnh trả về true/false | `(đầu ra) => đầu ra.length > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) - -🧩 Thiết lập MCP (Giao thức bối cảnh mô hình) +
+🧩 MCP Setup (Model Context Protocol) -Bắt đầu vận chuyển MCP ở chế độ stdio:```bash +Start MCP transport in stdio mode: + +```bash omniroute --mcp +``` -```` +Recommended validation flow: -Luồng xác thực được đề xuất: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. Kết nối máy khách MCP của bạn qua stdio. -2. Chạy `omniroute_get_health`. -3. Chạy `omniroute_list_combos`. -4. Mở `/dashboard/mcp` để xác nhận nhịp tim, hoạt động và kiểm tra. +Useful APIs for automation: -Các API hữu ích cho tự động hóa: +- `GET /api/mcp/status` +- `GET /api/mcp/tools` +- `GET /api/mcp/audit` +- `GET /api/mcp/audit/stats` -- `NHẬN/api/mcp/trạng thái` -- `NHẬN /api/mcp/công cụ` -- `NHẬN /api/mcp/kiểm toán` -- `NHẬN /api/mcp/kiểm toán/số liệu thống kê`
+ - -🤝 Thiết lập A2A (Agent2Agent) +
+🤝 A2A Setup (Agent2Agent) -Khám phá đại lý:```bash +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -Gửi một nhiệm vụ:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` +Manage lifecycle: -Quản lý vòng đời: - -- `NHẬN /api/a2a/trạng thái` -- `NHẬN /api/a2a/tác vụ` -- `NHẬN /api/a2a/tác vụ/:id` +- `GET /api/a2a/status` +- `GET /api/a2a/tasks` +- `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -Giao diện người dùng hoạt động: +Operational UI: -- `/dashboard/a2a` dành cho khả năng quan sát nhiệm vụ/trạng thái/luồng và các hành động khói
+- `/dashboard/a2a` for task/state/stream observability and smoke actions - -🧪 Xác thực giao thức end-to-end + -Xác thực cả hai giao thức với máy khách thực:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -Điều này xác minh: +This verifies: -- Kết nối/danh sách/cuộc gọi máy khách MCP SDK -- Khám phá/gửi/truyền phát/nhận/hủy A2A -- Kiểm tra chéo dữ liệu trong kiểm tra MCP và API quản lý tác vụ A2A
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs - -💳Nhà cung cấp dịch vụ đăng ký### Claude Code (Pro/Max) + + +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1403,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**Mẹo chuyên nghiệp:**Sử dụng Opus cho các tác vụ phức tạp, Sonnet cho tốc độ. OmniRoute theo dõi hạn ngạch cho mỗi mô hình!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1417,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -Mỗi tài khoản Codex hiện có các chuyển đổi chính sách trong `Bảng điều khiển -> Nhà cung cấp`: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h` (ON/OFF): thực thi chính sách ngưỡng cửa sổ 5 giờ. -- `Hàng tuần` (BẬT/TẮT): thực thi chính sách ngưỡng thời lượng hàng tuần. -- Hành vi ngưỡng: khi một cửa sổ được bật đạt mức sử dụng >=90%, tài khoản đó sẽ bị bỏ qua. -- Hành vi luân chuyển: OmniRoute tự động định tuyến đến tài khoản Codex đủ điều kiện tiếp theo. -- Hành vi đặt lại: khi thời gian `resetAt` của nhà cung cấp trôi qua, tài khoản sẽ tự động đủ điều kiện trở lại. +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -Kịch bản: +Scenarios: -- `5h ON` + `Weekly ON`: tài khoản bị bỏ qua khi một trong hai cửa sổ đạt đến ngưỡng. -- `TẮT 5h` + `BẬT hàng tuần`: chỉ sử dụng hàng tuần mới có thể khóa tài khoản. -- `5h BẬT` + `TẮT Hàng Tuần`: chỉ sử dụng 5h là có thể khóa tài khoản. -- `resetAt` đã qua: tài khoản tự động nhập lại vòng quay (không kích hoạt lại thủ công).### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1442,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**Giá trị tốt nhất:**Cấp miễn phí rất lớn! Sử dụng điều này trước các bậc trả phí.### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1457,73 +1662,91 @@ Models:
- +
+🔑 API Key Providers -🔑Nhà cung cấp khóa API### NVIDIA NIM (FREE developer access — 70+ models) +### NVIDIA NIM (FREE developer access — 70+ models) -1. Đăng ký: [build.nvidia.com](https://build.nvidia.com) -2. Nhận khóa API miễn phí (bao gồm 1000 tín dụng suy luận) -3. Bảng điều khiển → Thêm nhà cung cấp → NVIDIA NIM: - - Khóa API: `nvapi-your-key` +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**Mô hình:**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` và hơn 50 mẫu khác +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -**Mẹo chuyên nghiệp:**API tương thích với OpenAI — hoạt động trơn tru với tính năng dịch định dạng của OmniRoute!### DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -1. Đăng ký: [platform.deepseek.com](https://platform.deepseek.com) -2. Nhận khóa API -3. Trang tổng quan → Thêm nhà cung cấp → DeepSeek +### DeepSeek -**Mô hình:**`deepseek/deepseek-chat`, `deepseek/deepseek-code`### Groq (Free Tier Available!) +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -1. Đăng ký: [console.groq.com](https://console.groq.com) -2. Nhận khóa API (bao gồm bậc miễn phí) -3. Bảng điều khiển → Thêm nhà cung cấp → Groq +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**Mô hình:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b` +### Groq (Free Tier Available!) -**Mẹo chuyên nghiệp:**Suy luận cực nhanh — tốt nhất cho mã hóa thời gian thực!### OpenRouter (100+ Models) +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -1. Đăng ký: [openrouter.ai](https://openrouter.ai) -2. Nhận khóa API -3. Bảng điều khiển → Thêm nhà cung cấp → OpenRouter +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**Mô hình:**Truy cập hơn 100 mô hình từ tất cả các nhà cung cấp chính thông qua một khóa API duy nhất. +**Pro Tip:** Ultra-fast inference — best for real-time coding! -**Hoạt động của bảng điều khiển:**Các mô hình OpenRouter được quản lý từ**Các mô hình có sẵn**. Thêm, nhập và tự động đồng bộ hóa thủ công đều cập nhật cùng một danh sách.
+### OpenRouter (100+ Models) - -💰 Nhà cung cấp giá rẻ (Dự phòng)### GLM-4.7 (Daily reset, $0.6/1M) +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -1. Đăng ký: [Zhipu AI](https://open.bigmodel.cn/) -2. Nhận khóa API từ Gói mã hóa -3. Bảng điều khiển → Thêm khóa API: - - Nhà cung cấp: `glm` - - Khóa API: `your-key` +**Models:** Access 100+ models from all major providers through a single API key. -**Sử dụng:**`glm/glm-4.7` +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -**Mẹo chuyên nghiệp:**Gói mã hóa cung cấp hạn ngạch 3× với chi phí 1/7! Đặt lại vào 10:00 sáng hàng ngày.### MiniMax M2.1 (5h reset, $0.20/1M) + -1. Đăng ký: [MiniMax](https://www.minimax.io/) -2. Nhận khóa API -3. Bảng điều khiển → Thêm khóa API +
+💰 Cheap Providers (Backup) -**Sử dụng:**`minimax/MiniMax-M2.1` +### GLM-4.7 (Daily reset, $0.6/1M) -**Mẹo chuyên nghiệp:**Tùy chọn rẻ nhất cho ngữ cảnh dài (1 triệu mã thông báo)!### Kimi K2 ($9/month flat) +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -1. Đăng ký: [Moonshot AI](https://platform.moonshot.ai/) -2. Nhận khóa API -3. Bảng điều khiển → Thêm khóa API +**Use:** `glm/glm-4.7` -**Sử dụng:**`kimi/kimi-mới nhất` +**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. -**Mẹo chuyên nghiệp:**Đã sửa lỗi 9 USD/tháng cho 10 triệu mã thông báo = 0,90 USD/1 triệu chi phí hiệu quả!
+### MiniMax M2.1 (5h reset, $0.20/1M) - +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key -🆓 Nhà cung cấp MIỄN PHÍ (Dự phòng khẩn cấp)### Qoder (5 FREE models via OAuth) +**Use:** `minimax/MiniMax-M2.1` + +**Pro Tip:** Cheapest option for long context (1M tokens)! + +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1564,8 +1787,10 @@ Models:
- -🎨 Tạo Combo### Example 1: Maximize Subscription → Cheap Backup +
+🎨 Create Combos + +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1593,8 +1818,10 @@ Cost: $0 forever!
- -🔧 Tích hợp CLI### Cursor IDE +
+🔧 CLI Integration + +### Cursor IDE ``` Settings → Models → Advanced: @@ -1605,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -Sử dụng trang**Công cụ CLI**trong trang tổng quan để định cấu hình bằng một cú nhấp chuột hoặc chỉnh sửa `~/.claude/settings.json` theo cách thủ công.### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1616,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**Tùy chọn 1 — Trang tổng quan (được khuyến nghị):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**Tùy chọn 2 — Thủ công:**Chỉnh sửa `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1633,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **Lưu ý:**OpenClaw chỉ hoạt động với OmniRoute cục bộ. Sử dụng `127.0.0.1` thay vì `localhost` để tránh các vấn đề về độ phân giải IPv6.### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1647,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**Bước 1:**Thêm OmniRoute làm nhà cung cấp tùy chỉnh:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**Bước 2:**Tạo/chỉnh sửa `opencode.json` trong thư mục gốc dự án của bạn:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1673,117 +1909,130 @@ opencode } } } -```` +``` -**Bước 3:**Chọn model trong OpenCode:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**Mẹo:**Thêm bất kỳ mô hình nào có sẵn trong điểm cuối OmniRoute `/v1/models` của bạn vào phần `models`. Sử dụng định dạng `nhà cung cấp/model-id` từ bảng điều khiển OmniRoute của bạn.
+ --- ## Xử lý sự cố - -Nhấp để mở rộng hướng dẫn khắc phục sự cố +
+Click to expand troubleshooting guide -**"Mô hình ngôn ngữ không cung cấp tin nhắn"** +**"Language model did not provide messages"** - Provider quota exhausted → Check dashboard quota tracker -- Giải pháp: Sử dụng combo dự phòng hoặc chuyển sang tầng rẻ hơn +- Solution: Use combo fallback or switch to cheaper tier -**Giới hạn tỷ lệ** +**Rate limiting** -- Hết hạn ngạch đăng ký → Dự phòng sang GLM/MiniMax -- Thêm combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**Mã thông báo OAuth đã hết hạn** +**OAuth token expired** -- Tự động làm mới bởi OmniRoute -- Nếu sự cố vẫn tiếp diễn: Bảng điều khiển → Nhà cung cấp → Kết nối lại +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**Chi phí cao** +**High costs** -- Kiểm tra số liệu thống kê sử dụng trong Bảng điều khiển → Chi phí -- Chuyển mô hình chính sang GLM/MiniMax -- Sử dụng cấp miễn phí (Gemini CLI, Qoder) cho các tác vụ không quan trọng +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**Cổng bảng điều khiển/API bị sai** +**Dashboard/API ports are wrong** -- `PORT` là cổng cơ sở chuẩn (và cổng API theo mặc định) -- `API_PORT` chỉ ghi đè trình nghe API tương thích với OpenAI -- `DASHBOARD_PORT` chỉ ghi đè bảng điều khiển/trình nghe Next.js -- Đặt `NEXT_PUBLIC_BASE_URL` thành trang tổng quan/URL công khai của bạn (đối với lệnh gọi lại OAuth) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**Lỗi đồng bộ hóa đám mây** +**Cloud sync errors** -- Xác minh `BASE_URL` trỏ đến phiên bản đang chạy của bạn -- Xác minh `CLOUD_URL` trỏ đến điểm cuối đám mây dự kiến của bạn -- Giữ các giá trị `NEXT_PUBLIC_*` được căn chỉnh với các giá trị phía máy chủ +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Đăng nhập lần đầu không hoạt động** +**First login not working** -- Kiểm tra `INITIAL_PASSWORD` trong `.env` -- Nếu không được đặt, mật khẩu dự phòng là `123456` +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**Không có nhật ký yêu cầu** +**No request logs** -- Các tạo phẩm yêu cầu được ghi vào `DATA_DIR/call_logs/` dưới dạng một tệp JSON cho mỗi yêu cầu -- Kích hoạt tính năng chụp đường ống từ Bảng điều khiển → Nhật ký → Nhật ký yêu cầu nếu bạn cần tải trọng chi tiết theo từng giai đoạn -- Đặt `APP_LOG_TO_FILE=true` nếu bạn cũng muốn nhật ký bảng điều khiển ứng dụng trong `logs/application/app.log` -- Điều chỉnh `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` và `CALL_LOG_MAX_ENTRIES` nếu cần +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**Kiểm tra kết nối cho thấy "Không hợp lệ" đối với các nhà cung cấp tương thích với OpenAI** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- Nhiều nhà cung cấp không hiển thị điểm cuối `/models` -- OmniRoute v1.0.6+ bao gồm xác thực dự phòng thông qua hoàn thành trò chuyện -- Đảm bảo URL cơ sở bao gồm hậu tố `/v1`### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ Quan trọng đối với người dùng chạy OmniRoute trên VPS, Docker hoặc bất kỳ máy chủ từ xa nào**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -Các nhà cung cấp**AntiGravity**và**Gemini CLI**sử dụng**Google OAuth 2.0**. Google yêu cầu `redirect_uri` trong luồng OAuth phải khớp chính xác với một trong các URI đã đăng ký trước trong Google Cloud Console của ứng dụng. +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -Thông tin xác thực OAuth đi kèm trong OmniRoute được đăng ký**chỉ dành cho `localhost`**. Khi bạn truy cập OmniRoute trên máy chủ từ xa (ví dụ: `https://omniroute.myserver.com`), Google sẽ từ chối xác thực bằng:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -Bạn cần tạo**ID ứng dụng khách OAuth 2.0**trong Google Cloud Console bằng URI máy chủ của bạn.#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1. Mở Bảng điều khiển đám mây của Google** +#### Step-by-step -Truy cập: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2. Tạo ID ứng dụng khách OAuth 2.0 mới** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- Nhấp vào**"+ Tạo thông tin xác thực"**→**"ID khách hàng OAuth"** -- Loại ứng dụng:**"Ứng dụng web"** -- Tên: bất cứ thứ gì bạn thích (ví dụ: `OmniRoute Remote`) +**2. Create a new OAuth 2.0 Client ID** -**3. Thêm URI chuyển hướng được ủy quyền** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -Trong trường**"URI chuyển hướng được ủy quyền"**, hãy thêm:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> Thay thế `your-server.com` bằng miền hoặc IP máy chủ của bạn (bao gồm cổng nếu cần, ví dụ: `http://45.33.32.156:20128/callback`). +**4. Save and copy the credentials** -**4. Lưu và sao chép thông tin đăng nhập** +After creating, Google will show the **Client ID** and **Client Secret**. -Sau khi tạo, Google sẽ hiển thị**ID khách hàng**và**Bí mật khách hàng**. +**5. Set environment variables** -**5. Đặt biến môi trường** +In your `.env` (or Docker environment variables): -Trong `.env` (hoặc biến môi trường Docker của bạn):```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1792,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6. Khởi động lại OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7. Hãy thử kết nối lại** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -Bảng điều khiển → Nhà cung cấp → Anti Gravity (hoặc Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Giờ đây, Google sẽ chuyển hướng chính xác đến `https://your-server.com/callback`.--- +--- #### Temporary workaround (without custom credentials) -Nếu không muốn thiết lập thông tin xác thực của riêng mình ngay bây giờ, bạn vẫn có thể sử dụng**luồng URL thủ công**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1. OmniRoute mở URL ủy quyền của Google -2. Sau khi ủy quyền, Google sẽ cố gắng chuyển hướng đến `localhost` (không thành công trên máy chủ từ xa) -3.**Sao chép URL đầy đủ**từ thanh địa chỉ trình duyệt của bạn (ngay cả khi trang không tải) -4. Dán URL đó vào trường hiển thị trong phương thức kết nối OmniRoute -5. Nhấp vào**"Kết nối"** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> Điều này có tác dụng vì mã ủy quyền trong URL hợp lệ bất kể trang chuyển hướng có được tải hay không.--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. - -🇧🇷 Phiên bản tiếng Bồ Đào Nha#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -Os đã được chứng minh**AntiGravity**e**Gemini CLI**sử dụng**Google OAuth 2.0**để xác thực. O Google exige que a `redirect_uri` usada no fluxo OAuth seja**exatamente**một trong các URI trước cadastradas no Google Cloud Console để ứng dụng. +
+🇧🇷 Versão em Português -Vì các thông tin xác thực OAuth không có OmniRoute estão cadastradas**apenas para `localhost`**. Bạn có thể truy cập OmniRoute trong một thiết bị điều khiển lại máy chủ (ví dụ: `https://omniroute.meuservidor.com`), hoặc Google sẽ cung cấp thông tin xác thực bằng cách:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -Bạn nên chú ý**OAuth 2.0 Client ID**không có Google Cloud Console vì URI làm dịch vụ của bạn.#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1. Truy cập vào Google Cloud Console** +#### Passo a passo + +**1. Acesse o Google Cloud Console** Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -**2. Hãy yêu cầu ID khách hàng OAuth 2.0 mới** +**2. Crie um novo OAuth 2.0 Client ID** -- Nhấn vào**"+ Tạo thông tin xác thực"**→**"ID khách hàng OAuth"** -- Mẹo ứng dụng:**"Ứng dụng web"** -- Tên: escolha qualquer nome (ví dụ: `OmniRoute Remote`) +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -**3. Adicione dưới dạng URI chuyển hướng được ủy quyền** +**3. Adicione as Authorized Redirect URIs** -Không có quảng cáo**"URI chuyển hướng được ủy quyền"**, khuyến cáo:``` +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> Thay thế `seu-servidor.com` bằng địa chỉ IP hoặc IP của dịch vụ đó (bao gồm cổng bạn cần, ví dụ: `http://45.33.32.156:20128/callback`). +**4. Salve e copie as credenciais** -**4. Lưu và sao chép dưới dạng uy tín** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -Sau đó, Google đã đăng trên**ID khách hàng**và**Bí mật khách hàng**. +**5. Configure as variáveis de ambiente** -**5. Định cấu hình theo các biến thể của môi trường** +No seu `.env` (ou nas variáveis de ambiente do Docker): -Không có `.env` (hoặc có nhiều biến thể xung quanh Docker):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1871,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6. Reinicie hoặc OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute +``` -```` +**7. Tente conectar novamente** -**7. Lều kết nối mới lạ** +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -Bảng điều khiển → Nhà cung cấp → Anti Gravity (hoặc Gemini CLI) → OAuth +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. -Agora hoặc Google chuyển hướng điều chỉnh cho `https://seu-servidor.com/callback` và một chức năng xác thực.--- +--- #### Workaround temporário (sem configurar credenciais próprias) -Nếu không có câu hỏi nào về thông tin xác thực trước đây, bạn có thể sử dụng thông tin**hướng dẫn sử dụng URL**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. O OmniRoute tìm kiếm URL tự động của Google -2. Sau khi bạn tự động đăng ký, hoặc chuyển hướng Google sang `localhost` (que falha no servidor remoto) -3.**Sao chép một URL hoàn chỉnh**da cuối cùng của trình duyệt của bạn (mesmo que a página not carregue) -4. URL này không có khả năng xuất hiện không có phương thức kết nối nào với OmniRoute -5. Kết nối với nhau**"Kết nối"** +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) +4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute +5. Clique em **"Connect"** -> Chức năng giải pháp này có thể giúp tự động cấp quyền cho URL và có thể chuyển hướng độc lập đến mục tiêu hoặc không.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1909,64 +2171,73 @@ Nếu không có câu hỏi nào về thông tin xác thực trước đây, b ## 🛠️ Tech Stack - -Nhấp để xem chi tiết về ngăn xếp công nghệ +
+Click to expand tech stack details --**Thời gian chạy**: Node.js 18–22 LTS (⚠️ Node.js 24+**không được hỗ trợ**— các tệp nhị phân gốc `better-sqlite3` không tương thích) --**Ngôn ngữ**: TypeScript 5.9 —**100% TypeScript**trên `src/` và `open-sse/` (không có `any` trong các mô-đun lõi kể từ phiên bản 2.0) --**Khung**: Next.js 16 + React 19 + Tailwind CSS 4 --**Cơ sở dữ liệu**: LowDB (JSON) + SQLite (trạng thái miền + nhật ký proxy + kiểm tra MCP + quyết định định tuyến) --**Lược đồ**: Zod (xác thực I/O công cụ MCP, hợp đồng API) --**Giao thức**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**Truyền phát**: Sự kiện do máy chủ gửi (SSE) --**Xác thực**: OAuth 2.0 (PKCE) + JWT + Khóa API + Ủy quyền trong phạm vi MCP --**Thử nghiệm**: Trình chạy thử nghiệm Node.js + Vitest (hơn 900 thử nghiệm bao gồm đơn vị, tích hợp, E2E) --**CI/CD**: GitHub Actions (tự động xuất bản npm + Docker Hub khi phát hành) --**Trang web**: [omniroute.online](https://omniroute.online) --**Gói**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**Khả năng phục hồi**: Ngắt mạch, lùi theo cấp số nhân, chống sấm sét bầy đàn, giả mạo TLS, tự động kết hợp tự phục hồi
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## Tài liệu -| Tài liệu | Mô tả | +| Document | Description | | ---------------------------------------------- | --------------------------------------------------- | -| [Hướng dẫn sử dụng](docs/USER_GUIDE.md) | Nhà cung cấp, combo, tích hợp CLI, triển khai | -| [Tham khảo API](docs/API_REFERENCE.md) | Tất cả các điểm cuối có ví dụ | -| [Máy chủ MCP](open-sse/mcp-server/README.md) | 16 công cụ MCP, cấu hình IDE, máy khách Python/TS/Go | -| [Máy chủ A2A](src/lib/a2a/README.md) | Giao thức JSON-RPC 2.0, kỹ năng, phát trực tuyến, quản lý tác vụ | -| [Công cụ tự động kết hợp](docs/auto-combo.md) | Chấm điểm 6 yếu tố, gói chế độ, tự phục hồi | -| [Khắc phục sự cố](docs/TROUBLESHOOTING.md) | Các vấn đề thường gặp và giải pháp | -| [Kiến trúc](docs/ARCHITECTURE.md) | Kiến trúc hệ thống và nội bộ | -| [Đóng góp](CONTRIBUTING.md) | Thiết lập và hướng dẫn phát triển | -| [Thông số OpenAPI](docs/openapi.yaml) | Đặc tả OpenAPI 3.0 | -| [Chính sách bảo mật](SECURITY.md) | Báo cáo lỗ hổng bảo mật và thực hành bảo mật | -| [Triển khai VM](docs/VM_DEPLOYMENT_GUIDE.md) | Hướng dẫn đầy đủ: Thiết lập VM + nginx + Cloudflare | -| [Thư viện tính năng](docs/FEATURES.md) | Tham quan bảng điều khiển trực quan với ảnh chụp màn hình | -| [Danh sách kiểm tra bản phát hành](docs/RELEASE_CHECKLIST.md) | Các bước xác thực trước khi phát hành |--- +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute có**210+ tính năng được lên kế hoạch**qua nhiều giai đoạn phát triển. Dưới đây là các lĩnh vực chính: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -| Danh mục | Tính năng dự kiến ​​| Điểm nổi bật | +| Category | Planned Features | Highlights | | ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | -| 🧠**Định tuyến & thông minh**| 25+ | Định tuyến có độ trễ thấp nhất, định tuyến dựa trên thẻ, ưu tiên hạn ngạch, chọn tài khoản P2C | -| 🔒**Bảo mật & Tuân thủ**| 20+ | Tăng cường SSRF, che giấu thông tin xác thực, giới hạn tốc độ trên mỗi điểm cuối, phạm vi khóa quản lý | -| 📊**Khả năng quan sát**| 15+ | Tích hợp OpenTelemetry, giám sát hạn ngạch thời gian thực, theo dõi chi phí trên mỗi mô hình | -| 🔄**Tích hợp nhà cung cấp**| 20+ | Đăng ký mô hình động, thời gian hồi chiêu của nhà cung cấp, Codex nhiều tài khoản, phân tích hạn ngạch Copilot | -| ⚡**Hiệu suất**| 15+ | Lớp bộ đệm kép, bộ đệm nhắc nhở, bộ đệm phản hồi, lưu giữ luồng, API hàng loạt | -| 🌐**Hệ sinh thái**| 10+ | API WebSocket, cấu hình tải lại nóng, kho cấu hình phân tán, chế độ thương mại |### 🔜 Coming Soon +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**Tích hợp OpenCode**— Hỗ trợ của nhà cung cấp gốc cho IDE mã hóa OpenCode AI -- 🔗**Tích hợp TRAE**— Hỗ trợ đầy đủ cho khung phát triển TRAE AI -- 📦**Batch API**— Xử lý hàng loạt không đồng bộ cho các yêu cầu hàng loạt -- 🎯**Định tuyến dựa trên thẻ**— Định tuyến các yêu cầu dựa trên thẻ và siêu dữ liệu tùy chỉnh -- 💰**Chiến lược chi phí thấp nhất**— Tự động chọn nhà cung cấp có sẵn rẻ nhất +### 🔜 Coming Soon -> 📝 Thông số kỹ thuật đầy đủ của tính năng có sẵn trong [`docs/new-features/`](docs/new-features/) (217 thông số kỹ thuật chi tiết)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1974,18 +2245,20 @@ OmniRoute có**210+ tính năng được lên kế hoạch**qua nhiều giai đo ### How to Contribute -1. Phân nhánh kho lưu trữ -2. Tạo nhánh tính năng của bạn (`gitcheck -b feature/amazing-feature`) -3. Cam kết các thay đổi của bạn (`git commit -m 'Thêm tính năng tuyệt vời'`) -4. Đẩy tới nhánh (`git Push Origin feature/amazing-feature`) -5. Mở yêu cầu kéo +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -Xem [CONTRIBUTING.md](CONTRIBUTING.md) để biết hướng dẫn chi tiết.### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -1997,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -Đặc biệt cảm ơn**[9router](https://github.com/decolua/9router)**của**[decolua](https://github.com/decolua)**— dự án ban đầu đã truyền cảm hứng cho đợt phân nhánh này. OmniRoute được xây dựng dựa trên nền tảng đáng kinh ngạc đó với các tính năng bổ sung, API đa phương thức và viết lại TypeScript đầy đủ. +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -Đặc biệt cảm ơn**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— cách triển khai Go ban đầu đã truyền cảm hứng cho cổng JavaScript này.--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## Giấy phép -Giấy phép MIT - xem [LICENSE](LICENSE) để biết chi tiết.--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/vi/docs/ARCHITECTURE.md b/docs/i18n/vi/docs/ARCHITECTURE.md index 75aacf32bf..957da6a8b5 100644 --- a/docs/i18n/vi/docs/ARCHITECTURE.md +++ b/docs/i18n/vi/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_Cập nhật lần cuối: 28-03-2026_## Executive Summary -OmniRoute là cổng định tuyến và bảng thông tin AI cục bộ được xây dựng trên Next.js. -Nó cung cấp một điểm cuối tương thích với OpenAI (`/v1/*`) và định tuyến lưu lượng truy cập trên nhiều nhà cung cấp ngược dòng với bản dịch, dự phòng, làm mới mã thông báo và theo dõi việc sử dụng. -Khả năng cốt lõi: +_Last updated: 2026-03-28_ -- Bề mặt API tương thích OpenAI cho CLI/công cụ (28 nhà cung cấp) -- Dịch yêu cầu/phản hồi trên các định dạng của nhà cung cấp -- Dự phòng kết hợp mô hình (chuỗi nhiều mô hình) -- Dự phòng cấp tài khoản (nhiều tài khoản cho mỗi nhà cung cấp) -- Quản lý kết nối nhà cung cấp khóa OAuth + API -- Tạo nhúng thông qua `/v1/embeddings` (6 nhà cung cấp, 9 mô hình) -- Tạo hình ảnh qua `/v1/images/thế hệ` (4 nhà cung cấp, 9 kiểu máy) -- Phân tích thẻ suy nghĩ (`...`) cho các mô hình suy luận -- Dọn dẹp phản hồi để tương thích nghiêm ngặt với OpenAI SDK -- Chuẩn hóa vai trò (nhà phát triển→hệ thống, hệ thống→người dùng) để tương thích giữa các nhà cung cấp -- Chuyển đổi đầu ra có cấu trúc (json_schema → GeminiResponseSchema) -- Tính bền vững cục bộ cho nhà cung cấp, khóa, bí danh, tổ hợp, cài đặt, giá cả -- Theo dõi việc sử dụng/chi phí và ghi nhật ký yêu cầu -- Đồng bộ hóa đám mây tùy chọn để đồng bộ hóa nhiều thiết bị/trạng thái -- Danh sách cho phép/danh sách chặn IP để kiểm soát truy cập API -- Tư duy quản lý ngân sách (passthrough/auto/custom/adaptive) -- Tiêm nhắc nhở hệ thống toàn cầu -- Theo dõi phiên và lấy dấu vân tay -- Giới hạn tỷ lệ nâng cao cho mỗi tài khoản với hồ sơ dành riêng cho nhà cung cấp -- Mô hình ngắt mạch cho khả năng phục hồi của nhà cung cấp -- Bảo vệ đàn chống sét bằng khóa mutex -- Bộ đệm chống trùng lặp yêu cầu dựa trên chữ ký -- Lớp miền: tính khả dụng của mô hình, quy tắc chi phí, chính sách dự phòng, chính sách khóa -- Tính bền vững của trạng thái miền (bộ đệm ghi SQLite dành cho dự phòng, ngân sách, khóa, bộ ngắt mạch) -- Công cụ chính sách để đánh giá yêu cầu tập trung (khóa → ngân sách → dự phòng) -- Yêu cầu đo từ xa với tổng hợp độ trễ p50/p95/p99 -- ID tương quan (X-Request-Id) để theo dõi từ đầu đến cuối -- Ghi nhật ký kiểm tra tuân thủ với tính năng chọn không tham gia trên mỗi khóa API -- Khung đánh giá để đảm bảo chất lượng LLM -- Bảng điều khiển UI có khả năng phục hồi với trạng thái ngắt mạch theo thời gian thực -- Nhà cung cấp OAuth mô-đun (12 mô-đun riêng lẻ trong `src/lib/oauth/providers/`) +## Executive Summary -Mô hình thời gian chạy chính: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- Các tuyến ứng dụng Next.js trong `src/app/api/*` triển khai cả API bảng điều khiển và API tương thích -- Một lõi định tuyến/SSE được chia sẻ trong `src/sse/*` + `open-sse/*` xử lý việc thực thi, dịch thuật, phát trực tuyến, dự phòng và sử dụng của nhà cung cấp## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- Thời gian chạy cổng cục bộ -- API quản lý bảng điều khiển -- Xác thực nhà cung cấp và làm mới mã thông báo -- Yêu cầu dịch và truyền phát SSE -- Trạng thái cục bộ + kiên trì sử dụng -- Phối hợp đồng bộ hóa đám mây tùy chọn### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- Triển khai dịch vụ đám mây đằng sau `NEXT_PUBLIC_CLOUD_URL` -- Nhà cung cấp SLA/mặt phẳng điều khiển bên ngoài quy trình cục bộ -- Bản thân các tệp nhị phân CLI bên ngoài (Claude CLI, Codex CLI, v.v.)## Dashboard Surface (Current) +### Out of Scope -Các trang chính trong `src/app/(dashboard)/dashboard/`: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — bắt đầu nhanh + tổng quan về nhà cung cấp -- `/dashboard/endpoint` — proxy điểm cuối + MCP + A2A + tab điểm cuối API -- `/dashboard/providers` — kết nối và thông tin đăng nhập của nhà cung cấp -- `/dashboard/combos` — chiến lược kết hợp, mẫu, quy tắc định tuyến mô hình -- `/dashboard/costs` — tổng hợp chi phí và khả năng hiển thị giá cả -- `/dashboard/analytics` — phân tích và đánh giá việc sử dụng -- `/dashboard/limits` — kiểm soát hạn ngạch/tỷ lệ -- `/dashboard/cli-tools` — Tích hợp CLI, phát hiện thời gian chạy, tạo cấu hình -- `/dashboard/agents` — các tác nhân ACP được phát hiện + đăng ký tác nhân tùy chỉnh -- `/dashboard/media` — sân chơi hình ảnh/video/âm nhạc -- `/dashboard/search-tools` — lịch sử và kiểm tra nhà cung cấp dịch vụ tìm kiếm -- `/dashboard/health` — thời gian hoạt động, ngắt mạch, giới hạn tốc độ -- `/dashboard/logs` — nhật ký yêu cầu/proxy/kiểm toán/bàn điều khiển -- `/dashboard/settings` — các tab cài đặt hệ thống (chung, định tuyến, mặc định kết hợp, v.v.) -- `/dashboard/api-manager` — Quyền đối với mô hình và vòng đời của khóa API## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -Các thư mục chính: +Main directories: -- `src/app/api/v1/*` và `src/app/api/v1beta/*` cho các API tương thích -- `src/app/api/*` dành cho API quản lý/cấu hình -- Tiếp theo viết lại trong `next.config.mjs` ánh xạ `/v1/*` thành `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -Các tuyến tương thích quan trọng: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` — bao gồm các mô hình tùy chỉnh với `custom: true` -- `src/app/api/v1/embeddings/route.ts` — thế hệ nhúng (6 nhà cung cấp) -- `src/app/api/v1/images/Generations/route.ts` — tạo hình ảnh (4+ nhà cung cấp bao gồm AntiGravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — trò chuyện dành riêng cho mỗi nhà cung cấp -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — phần nhúng dành riêng cho mỗi nhà cung cấp -- `src/app/api/v1/providers/[provider]/images/thế hệ/route.ts` — hình ảnh dành riêng cho mỗi nhà cung cấp +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` - `src/app/api/v1beta/models/[...path]/route.ts` -Các miền quản lý: +Management domains: -- Xác thực/cài đặt: `src/app/api/auth/*`, `src/app/api/settings/*` -- Nhà cung cấp/kết nối: `src/app/api/providers*` -- Các nút của nhà cung cấp: `src/app/api/provider-nodes*` -- Model tùy chỉnh: `src/app/api/provider-models` (GET/POST/DELETE) -- Danh mục mô hình: `src/app/api/models/route.ts` (GET) -- Cấu hình proxy: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) - OAuth: `src/app/api/oauth/*` -- Khóa/bí danh/combos/giá: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` -- Cách sử dụng: `src/app/api/usage/*` -- Đồng bộ/đám mây: `src/app/api/sync/*`, `src/app/api/cloud/*` -- Trình trợ giúp công cụ CLI: `src/app/api/cli-tools/*` -- Bộ lọc IP: `src/app/api/settings/ip-filter` (GET/PUT) -- Ngân sách tư duy: `src/app/api/settings/thinking-budget` (GET/PUT) -- Dấu nhắc hệ thống: `src/app/api/settings/system-prompt` (GET/PUT) -- Phiên: `src/app/api/sessions` (GET) -- Giới hạn tỷ lệ: `src/app/api/rate-limits` (GET) -- Khả năng phục hồi: `src/app/api/resilience` (GET/PATCH) — hồ sơ nhà cung cấp, bộ ngắt mạch, trạng thái giới hạn tốc độ -- Đặt lại khả năng phục hồi: `src/app/api/resilience/reset` (POST) — đặt lại bộ ngắt + thời gian hồi chiêu -- Thống kê bộ đệm: `src/app/api/cache/stats` (GET/DELETE) -- Tính khả dụng của mô hình: `src/app/api/models/availability` (GET/POST) -- Đo từ xa: `src/app/api/telemetry/summary` (GET) -- Ngân sách: `src/app/api/usage/budget` (GET/POST) -- Chuỗi dự phòng: `src/app/api/fallback/chains` (GET/POST/DELETE) -- Kiểm tra tuân thủ: `src/app/api/compliance/audit-log` (GET) -- Đánh giá: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) -- Chính sách: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -Các mô-đun dòng chảy chính: +## 2) SSE + Translation Core -- Mục nhập: `src/sse/handlers/chat.ts` -- Điều phối cốt lõi: `open-sse/handlers/chatCore.ts` -- Bộ điều hợp thực thi của nhà cung cấp: `open-sse/executors/*` -- Phát hiện định dạng/cấu hình nhà cung cấp: `open-sse/services/provider.ts` -- Phân tích/giải quyết mô hình: `src/sse/services/model.ts`, `open-sse/services/model.ts` -- Logic dự phòng tài khoản: `open-sse/services/accountFallback.ts` -- Sổ đăng ký dịch: `open-sse/translator/index.ts` -- Chuyển đổi luồng: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` -- Trích xuất/chuẩn hóa cách sử dụng: `open-sse/utils/usageTracking.ts` -- Trình phân tích cú pháp thẻ Think: `open-sse/utils/thinkTagParser.ts` -- Trình xử lý nhúng: `open-sse/handlers/embeddings.ts` -- Đăng ký nhà cung cấp nhúng: `open-sse/config/embeddingRegistry.ts` -- Trình xử lý tạo hình ảnh: `open-sse/handlers/imageGeneration.ts` -- Sổ đăng ký nhà cung cấp hình ảnh: `open-sse/config/imageRegistry.ts` -- Làm sạch phản hồi: `open-sse/handlers/responseSanitizer.ts` -- Chuẩn hóa vai trò: `open-sse/services/roleNormalizer.ts` +Main flow modules: -Dịch vụ (logic nghiệp vụ): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- Lựa chọn/chấm điểm tài khoản: `open-sse/services/accountSelector.ts` -- Quản lý vòng đời bối cảnh: `open-sse/services/contextManager.ts` -- Thực thi bộ lọc IP: `open-sse/services/ipFilter.ts` -- Theo dõi phiên: `open-sse/services/sessionManager.ts` -- Yêu cầu loại bỏ trùng lặp: `open-sse/services/signatureCache.ts` -- Nội dung nhắc nhở của hệ thống: `open-sse/services/systemPrompt.ts` -- Tư duy quản lý ngân sách: `open-sse/services/thinkingBudget.ts` -- Định tuyến mô hình ký tự đại diện: `open-sse/services/wildcardRouter.ts` -- Quản lý giới hạn tỷ lệ: `open-sse/services/rateLimitManager.ts` -- Bộ ngắt mạch: `open-sse/services/ CircuitBreaker.ts` +Services (business logic): -Các mô-đun lớp miền: +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- Tính khả dụng của mô hình: `src/lib/domain/modelAvailability.ts` -- Quy tắc chi phí/ngân sách: `src/lib/domain/costRules.ts` -- Chính sách dự phòng: `src/lib/domain/fallbackPolicy.ts` -- Trình phân giải kết hợp: `src/lib/domain/comboResolver.ts` -- Chính sách khóa: `src/lib/domain/lockoutPolicy.ts` -- Công cụ chính sách: `src/domain/policyEngine.ts` — khóa tập trung → ngân sách → đánh giá dự phòng -- Danh mục mã lỗi: `src/lib/domain/errorCodes.ts` -- ID yêu cầu: `src/lib/domain/requestId.ts` -- Thời gian chờ tìm nạp: `src/lib/domain/fetchTimeout.ts` -- Yêu cầu đo từ xa: `src/lib/domain/requestTelemetry.ts` -- Tuân thủ/kiểm toán: `src/lib/domain/compliance/index.ts` -- Người chạy đánh giá: `src/lib/domain/evalRunner.ts` -- Tính bền vững của trạng thái miền: `src/lib/db/domainState.ts` — SQLite CRUD dành cho chuỗi dự phòng, ngân sách, lịch sử chi phí, trạng thái khóa, bộ ngắt mạch +Domain layer modules: -Mô-đun nhà cung cấp OAuth (12 tệp riêng lẻ trong `src/lib/oauth/providers/`): +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -- Chỉ mục đăng ký: `src/lib/oauth/providers/index.ts` -- Các nhà cung cấp cá nhân: `claude.ts`, `codex.ts`, `gemini.ts`, `antiGravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` -- Trình bao bọc mỏng: `src/lib/oauth/providers.ts` — tái xuất từ các mô-đun riêng lẻ## 3) Persistence Layer +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -DB trạng thái chính (SQLite): +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -- Cơ sở hạ tầng cốt lõi: `src/lib/db/core.ts` (tốt hơn-sqlite3, di chuyển, WAL) -- Mặt tiền tái xuất: `src/lib/localDb.ts` (lớp tương thích mỏng cho người gọi) -- tệp: `${DATA_DIR}/storage.sqlite` (hoặc `$XDG_CONFIG_HOME/omniroute/storage.sqlite` khi được đặt, nếu không thì `~/.omniroute/storage.sqlite`) -- thực thể (bảng + không gian tên KV): nhà cung cấpConnections, nhà cung cấpNodes, modelAliases, combo, apiKeys, cài đặt, giá cả,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt** +## 3) Persistence Layer -Kiên trì sử dụng: +Primary state DB (SQLite): -- mặt tiền: `src/lib/usageDb.ts` (các mô-đun được phân tách trong `src/lib/usage/*`) -- Bảng SQLite trong `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` -- các tạo phẩm tệp tùy chọn vẫn còn để tương thích/gỡ lỗi (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) -- các tệp JSON kế thừa được di chuyển sang SQLite bằng cách di chuyển khởi động khi có mặt +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -Cơ sở dữ liệu trạng thái miền (SQLite): +Usage persistence: -- `src/lib/db/domainState.ts` — Hoạt động CRUD cho trạng thái miền -- Các bảng (được tạo trong `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circle_breakers` -- Mẫu bộ đệm ghi qua: Bản đồ trong bộ nhớ có thẩm quyền trong thời gian chạy; các đột biến được ghi đồng bộ vào SQLite; trạng thái được khôi phục từ DB khi khởi động nguội## 4) Auth + Security Surfaces +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- Xác thực cookie bảng điều khiển: `src/proxy.ts`, `src/app/api/auth/login/route.ts` -- Tạo/xác minh khóa API: `src/shared/utils/apiKey.ts` -- Bí mật của nhà cung cấp vẫn tồn tại trong các mục `providerConnections` -- Hỗ trợ proxy gửi đi thông qua `open-sse/utils/proxyFetch.ts` (env vars) và `open-sse/utils/networkProxy.ts` (có thể định cấu hình cho mỗi nhà cung cấp hoặc toàn cầu)## 5) Cloud Sync +Domain State DB (SQLite): -- Trình lập lịch biểu khởi tạo: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` -- Nhiệm vụ định kỳ: `src/shared/services/cloudSyncScheduler.ts` -- Tác vụ định kỳ: `src/shared/services/modelSyncScheduler.ts` -- Tuyến điều khiển: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -Các quyết định dự phòng được điều khiển bởi `open-sse/services/accountFallback.ts` bằng cách sử dụng mã trạng thái và phương pháp phỏng đoán thông báo lỗi. Định tuyến kết hợp bổ sung thêm một biện pháp bảo vệ: 400 lỗi trong phạm vi nhà cung cấp, chẳng hạn như lỗi xác thực vai trò và khối nội dung ngược dòng được coi là lỗi cục bộ mô hình để các mục tiêu kết hợp sau này vẫn có thể chạy.## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -Làm mới trong khi lưu lượng truy cập trực tiếp được thực thi bên trong `open-sse/handlers/chatCore.ts` thông qua trình thực thi `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -Đồng bộ hóa định kỳ được kích hoạt bởi `CloudSyncScheduler` khi bật đám mây.## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -Tệp lưu trữ vật lý: +Physical storage files: -- DB thời gian chạy chính: `${DATA_DIR}/storage.sqlite` -- dòng nhật ký yêu cầu: `${DATA_DIR}/log.txt` (tạo phẩm tương thích/gỡ lỗi) -- kho lưu trữ tải trọng cuộc gọi có cấu trúc: `${DATA_DIR}/call_logs/` -- phiên gỡ lỗi yêu cầu/trình dịch tùy chọn: `/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,205 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`, `src/app/api/v1beta/*`: các API tương thích -- `src/app/api/v1/providers/[provider]/*`: các tuyến dành riêng cho mỗi nhà cung cấp (trò chuyện, nhúng, hình ảnh) -- `src/app/api/providers*`: CRUD của nhà cung cấp, xác thực, kiểm tra -- `src/app/api/provider-nodes*`: quản lý nút tương thích tùy chỉnh -- `src/app/api/provider-models`: quản lý mô hình tùy chỉnh (CRUD) -- `src/app/api/models/route.ts`: API danh mục mô hình (bí danh + mô hình tùy chỉnh) -- `src/app/api/oauth/*`: Luồng OAuth/mã thiết bị -- `src/app/api/keys*`: vòng đời của khóa API cục bộ -- `src/app/api/models/alias`: quản lý bí danh -- `src/app/api/combos*`: quản lý kết hợp dự phòng -- `src/app/api/pricing`: ghi đè giá để tính chi phí -- `src/app/api/settings/proxy`: cấu hình proxy (GET/PUT/DELETE) +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) - `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) -- `src/app/api/usage/*`: API sử dụng và nhật ký -- `src/app/api/sync/*` + `src/app/api/cloud/*`: đồng bộ hóa đám mây và trợ giúp đối mặt với đám mây -- `src/app/api/cli-tools/*`: trình soạn thảo/kiểm tra cấu hình CLI cục bộ -- `src/app/api/settings/ip-filter`: Danh sách cho phép/danh sách chặn IP (GET/PUT) -- `src/app/api/settings/thinking-budget`: cấu hình ngân sách mã thông báo suy nghĩ (GET/PUT) -- `src/app/api/settings/system-prompt`: dấu nhắc hệ thống toàn cầu (GET/PUT) -- `src/app/api/sessions`: danh sách phiên hoạt động (GET) -- `src/app/api/rate-limits`: trạng thái giới hạn tỷ lệ cho mỗi tài khoản (GET)### Routing and Execution Core +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`: phân tích cú pháp yêu cầu, xử lý kết hợp, vòng lặp chọn tài khoản -- `open-sse/handlers/chatCore.ts`: dịch, gửi người thực thi, xử lý thử lại/làm mới, thiết lập luồng -- `open-sse/executors/*`: hành vi định dạng và mạng dành riêng cho nhà cung cấp### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`: đăng ký và điều phối dịch giả -- Yêu cầu người dịch: `open-sse/translator/request/*` -- Trình dịch phản hồi: `open-sse/translator/response/*` -- Hằng định dạng: `open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`: duy trì cấu hình/trạng thái và tên miền liên tục trên SQLite -- `src/lib/localDb.ts`: tái xuất khả năng tương thích cho các mô-đun DB -- `src/lib/usageDb.ts`: mặt tiền lịch sử sử dụng/nhật ký cuộc gọi ở đầu các bảng SQLite## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -Mỗi nhà cung cấp có một trình thực thi chuyên biệt mở rộng `BaseExecutor` (trong `open-sse/executors/base.ts`), cung cấp việc xây dựng URL, xây dựng tiêu đề, thử lại với thời gian chờ theo cấp số nhân, móc làm mới thông tin xác thực và phương thức điều phối `execute()`. +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| Người thi hành | (Các) nhà cung cấp | Xử lý đặc biệt | -| ------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------- | -| `Trình thực thi mặc định` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Cấu hình URL/tiêu đề động cho mỗi nhà cung cấp | -| `Người thực thi phản trọng lực` | Google phản lực hấp dẫn | ID dự án/phiên tùy chỉnh, Thử lại sau khi phân tích cú pháp | -| `CodexExecutor` | OpenAI Codex | Đưa vào các hướng dẫn hệ thống, buộc nỗ lực suy luận | -| `Người thực thi con trỏ` | IDE con trỏ | Giao thức ConnectRPC, mã hóa Protobuf, ký yêu cầu qua tổng kiểm tra | -| `GithubExecutor` | Phi công phụ GitHub | Làm mới mã thông báo Copilot, tiêu đề bắt chước VSCode | -| `KiroExecutor` | AWS CodeWhisperer/Kiro | Định dạng nhị phân AWS EventStream → Chuyển đổi SSE | -| `GeminiCLIExecutor` | Song Tử CLI | Chu kỳ làm mới mã thông báo Google OAuth | +### Persistence -Tất cả các nhà cung cấp khác (bao gồm các nút tương thích tùy chỉnh) đều sử dụng `DefaultExecutor`.## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| Nhà cung cấp | Định dạng | Xác thực | Truyền phát | Không phát trực tuyến | Làm mới mã thông báo | API sử dụng | -| ------------------- | --------------- | ----------------------------- | ---------------- | --------------------- | -------------------- | ----------------------------- | ------------------------------ | -| Claude | Claude | Khóa API / OAuth | ✅ | ✅ | ✅ | ⚠️ Chỉ dành cho quản trị viên | -| Song Tử | song tử | Khóa API / OAuth | ✅ | ✅ | ✅ | ⚠️ Bảng điều khiển đám mây | -| Song Tử CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Bảng điều khiển đám mây | -| Phản lực hấp dẫn | phản trọng lực | OAuth | ✅ | ✅ | ✅ | ✅ API hạn ngạch đầy đủ | -| OpenAI | mở | Khóa API | ✅ | ✅ | ❌ | ❌ | -| Codex | phản hồi openai | OAuth | ✅ ép buộc | ❌ | ✅ | ✅ Giới hạn tỷ lệ | -| Phi công phụ GitHub | mở | OAuth + Mã thông báo đồng lái | ✅ | ✅ | ✅ | ✅ Ảnh chụp nhanh hạn ngạch | -| Con trỏ | con trỏ | Tổng kiểm tra tùy chỉnh | ✅ | ✅ | ❌ | ❌ | -| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Giới hạn sử dụng | -| Qwen | mở | OAuth | ✅ | ✅ | ✅ | ⚠️ Theo yêu cầu | -| Qoder | mở | OAuth (Cơ bản) | ✅ | ✅ | ✅ | ⚠️ Theo yêu cầu | -| OpenRouter | mở | Khóa API | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | Claude | Khóa API | ✅ | ✅ | ❌ | ❌ | -| DeepSeek | mở | Khóa API | ✅ | ✅ | ❌ | ❌ | -| Groq | mở | Khóa API | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | mở | Khóa API | ✅ | ✅ | ❌ | ❌ | -| Mistral | mở | Khóa API | ✅ | ✅ | ❌ | ❌ | -| Lúng túng | mở | Khóa API | ✅ | ✅ | ❌ | ❌ | -| Cùng AI | mở | Khóa API | ✅ | ✅ | ❌ | ❌ | -| Pháo hoa AI | mở | Khóa API | ✅ | ✅ | ❌ | ❌ | -| Não | mở | Khóa API | ✅ | ✅ | ❌ | ❌ | -| Kết hợp | mở | Khóa API | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | mở | Khóa API | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -Các định dạng nguồn được phát hiện bao gồm: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. + +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | + +All other providers (including custom compatible nodes) use the `DefaultExecutor`. + +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: - `openai` -- `openai-phản hồi` -- `claudia` -- `song tử` +- `openai-responses` +- `claude` +- `gemini` -Các định dạng mục tiêu bao gồm: +Target formats include: -- Trò chuyện/Phản hồi OpenAI +- OpenAI chat/Responses - Claude -- Phong bì Gemini/Gemini-CLI/Phản trọng lực +- Gemini/Gemini-CLI/Antigravity envelope - Kiro -- Con trỏ +- Cursor -Các bản dịch sử dụng**OpenAI làm định dạng trung tâm**— tất cả các chuyển đổi đều thông qua OpenAI dưới dạng trung gian:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -Các bản dịch được chọn linh hoạt dựa trên hình dạng tải trọng nguồn và định dạng mục tiêu của nhà cung cấp. +Additional processing layers in the translation pipeline: -Các lớp xử lý bổ sung trong quy trình dịch thuật: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**Sạch hóa phản hồi**— Loại bỏ các trường không chuẩn khỏi phản hồi ở định dạng OpenAI (cả phát trực tuyến và không phát trực tuyến) để đảm bảo tuân thủ nghiêm ngặt SDK --**Chuẩn hóa vai trò**— Chuyển đổi `developer` → `system` cho các mục tiêu không phải OpenAI; hợp nhất `system` → `user` cho các mô hình từ chối vai trò hệ thống (GLM, ERNIE) --**Trích xuất thẻ Think**— Phân tích cú pháp `...` chặn nội dung vào trường `reasoning_content` --**Đầu ra có cấu trúc**— Chuyển đổi `response_format.json_schema` của OpenAI thành `responseMimeType` + `responseSchema` của Gemini## Supported API Endpoints +## Supported API Endpoints -| Điểm cuối | Định dạng | Người xử lý | +| Endpoint | Format | Handler | | -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | -| `POST /v1/chat/hoàn thành` | Trò chuyện OpenAI | `src/sse/handlers/chat.ts` | -| `POST /v1/tin nhắn` | Tin nhắn Claude | Trình xử lý tương tự (tự động phát hiện) | -| `POST /v1/phản hồi` | Phản hồi OpenAI | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/nhúng` | Nhúng OpenAI | `open-sse/handlers/embeddings.ts` | -| `NHẬN /v1/nhúng` | Danh sách mô hình | Tuyến đường API | -| `POST /v1/images/thế hệ` | Hình ảnh OpenAI | `open-sse/handlers/imageGeneration.ts` | -| `NHẬN /v1/hình ảnh/thế hệ` | Danh sách mô hình | Tuyến đường API | -| `POST /v1/providers/{provider}/chat/completions` | Trò chuyện OpenAI | Dành riêng cho mỗi nhà cung cấp với xác thực mô hình | -| `POST /v1/providers/{provider}/embeddings` | Nhúng OpenAI | Dành riêng cho mỗi nhà cung cấp với xác thực mô hình | -| `POST /v1/providers/{provider}/images/thế hệ` | Hình ảnh OpenAI | Dành riêng cho mỗi nhà cung cấp với xác thực mô hình | -| `POST /v1/messages/count_tokens` | Số lượng mã thông báo Claude | Tuyến đường API | -| `NHẬN /v1/model` | Danh sách mô hình OpenAI | Tuyến API (trò chuyện + nhúng + hình ảnh + mô hình tùy chỉnh) | -| `NHẬN /api/mô hình/danh mục` | Danh mục | Tất cả các mô hình được nhóm theo nhà cung cấp + loại | -| `POST /v1beta/models/*:streamGenerateContent` | Song Tử bản xứ | Tuyến đường API | -| `NHẬN/PUT/XÓA /api/settings/proxy` | Cấu hình proxy | Cấu hình proxy mạng | -| `POST /api/settings/proxy/test` | Kết nối proxy | Điểm cuối kiểm tra sức khỏe/kết nối proxy | -| `GET/POST/DELETE /api/provider-models` | Mô hình nhà cung cấp | Sao lưu siêu dữ liệu mô hình nhà cung cấp các mô hình có sẵn tùy chỉnh và được quản lý |## Bypass Handler +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -Trình xử lý bỏ qua (`open-sse/utils/bypassHandler.ts`) chặn các yêu cầu "loại bỏ" đã biết từ Claude CLI — ping khởi động, trích xuất tiêu đề và số lượng mã thông báo — và trả về**phản hồi giả**mà không tiêu tốn mã thông báo của nhà cung cấp ngược dòng. Điều này chỉ được kích hoạt khi `User-Agent` chứa `claude-cli`.## Request Logger Pipeline +## Bypass Handler -Trình ghi nhật ký yêu cầu (`open-sse/utils/requestLogger.ts`) cung cấp quy trình ghi nhật ký gỡ lỗi 7 giai đoạn, bị tắt theo mặc định, được bật thông qua `ENABLE_REQUEST_LOGS=true`:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -Các tập tin được ghi vào `/logs//` cho mỗi phiên yêu cầu.## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- thời gian hồi chiêu của tài khoản nhà cung cấp đối với các lỗi tạm thời/tỷ lệ/xác thực -- dự phòng tài khoản trước khi yêu cầu không thành công -- dự phòng mô hình kết hợp khi đường dẫn mô hình/nhà cung cấp hiện tại đã hết## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- kiểm tra trước và làm mới bằng cách thử lại đối với các nhà cung cấp có thể làm mới -- Thử lại 401/403 sau lần thử làm mới trong đường dẫn lõi## 3) Stream Safety +## 2) Token Expiry -- bộ điều khiển luồng nhận biết ngắt kết nối -- luồng dịch với tính năng xả cuối luồng và xử lý `[DONE]` -- dự phòng ước tính sử dụng khi thiếu siêu dữ liệu sử dụng của nhà cung cấp## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- lỗi đồng bộ hóa xuất hiện nhưng thời gian chạy cục bộ vẫn tiếp tục -- bộ lập lịch có logic có khả năng thử lại, nhưng việc thực thi định kỳ hiện gọi đồng bộ hóa một lần thử theo mặc định## 5) Data Integrity +## 3) Stream Safety -- Di chuyển lược đồ SQLite và móc nâng cấp tự động khi khởi động -- JSON kế thừa → Đường dẫn tương thích di chuyển SQLite## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -Nguồn hiển thị thời gian chạy: +## 4) Cloud Sync Degradation -- nhật ký bảng điều khiển từ `src/sse/utils/logger.ts` -- tổng hợp mức sử dụng theo yêu cầu trong SQLite (`usage_history`, `call_logs`, `proxy_logs`) -- Ghi lại tải trọng chi tiết bốn giai đoạn trong SQLite (`request_detail_logs`) khi `settings.detailed_logs_enabled=true` -- nhật ký trạng thái yêu cầu văn bản trong `log.txt` (tùy chọn/tương thích) -- nhật ký dịch/yêu cầu sâu tùy chọn trong `logs/` khi `ENABLE_REQUEST_LOGS=true` -- điểm cuối sử dụng bảng điều khiển (`/api/usage/*`) để sử dụng giao diện người dùng +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -Tính năng thu thập tải trọng yêu cầu chi tiết lưu trữ tối đa bốn giai đoạn tải trọng JSON cho mỗi cuộc gọi định tuyến: +## 5) Data Integrity -- yêu cầu thô nhận được từ khách hàng -- yêu cầu đã dịch thực sự được gửi ngược dòng -- phản hồi của nhà cung cấp được xây dựng lại dưới dạng JSON; phản hồi theo luồng được nén thành bản tóm tắt cuối cùng cộng với siêu dữ liệu luồng -- phản hồi cuối cùng của khách hàng được OmniRoute trả về; các phản hồi theo luồng được lưu trữ ở cùng một dạng tóm tắt nhỏ gọn## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- Bí mật JWT (`JWT_SECRET`) bảo mật việc xác minh/ký cookie phiên bảng điều khiển -- Khởi động mật khẩu ban đầu (`INITIAL_PASSWORD`) phải được định cấu hình rõ ràng để cung cấp lần đầu -- Khóa API Bí mật HMAC (`API_KEY_SECRET`) bảo mật định dạng khóa API cục bộ được tạo -- Bí mật của nhà cung cấp (khóa API/mã thông báo) được lưu giữ trong DB cục bộ và phải được bảo vệ ở cấp hệ thống tệp -- Điểm cuối đồng bộ hóa đám mây dựa vào ngữ nghĩa xác thực khóa API + id máy## Environment and Runtime Matrix +## Observability and Operational Signals -Các biến môi trường được mã sử dụng tích cực: +Runtime visibility sources: -- Ứng dụng/xác thực: `JWT_SECRET`, `INITIAL_PASSWORD` -- Bộ nhớ: `DATA_DIR` -- Hành vi của nút tương thích: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- Ghi đè cơ sở lưu trữ tùy chọn (Linux/macOS khi không đặt `DATA_DIR`): `XDG_CONFIG_HOME` -- Băm bảo mật: `API_KEY_SECRET`, `MACHINE_ID_SALT` -- Ghi nhật ký: `ENABLE_REQUEST_LOGS` -- URL đồng bộ hóa/đám mây: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` -- Proxy gửi đi: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` và các biến thể chữ thường -- Cờ tính năng SOCKS5: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- Trình trợ giúp nền tảng/thời gian chạy (không phải cấu hình dành riêng cho ứng dụng): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` và `localDb` chia sẻ cùng một chính sách thư mục cơ sở (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) với việc di chuyển tệp kế thừa. -2. `/api/v1/route.ts` ủy quyền cho cùng một trình tạo danh mục hợp nhất được sử dụng bởi `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) để tránh trôi dạt ngữ nghĩa. -3. Trình ghi yêu cầu ghi toàn bộ tiêu đề/nội dung khi được bật; coi thư mục nhật ký là nhạy cảm. -4. Hoạt động của đám mây phụ thuộc vào `NEXT_PUBLIC_BASE_URL` chính xác và khả năng tiếp cận điểm cuối của đám mây. -5. Thư mục `open-sse/` được xuất bản dưới dạng `@omniroute/open-sse`**gói không gian làm việc npm**. Mã nguồn nhập nó qua `@omniroute/open-sse/...` (được giải quyết bởi Next.js `transpilePackages`). Đường dẫn tệp trong tài liệu này vẫn sử dụng tên thư mục `open-sse/` để đảm bảo tính thống nhất. -6. Các biểu đồ trong trang tổng quan sử dụng**Recharts**(dựa trên SVG) để hiển thị trực quan hóa phân tích tương tác, có thể truy cập (biểu đồ thanh sử dụng mô hình, bảng phân tích nhà cung cấp với tỷ lệ thành công). -7. Kiểm thử E2E sử dụng**Playwright**(`tests/e2e/`), chạy qua `npm run test:e2e`. Kiểm thử đơn vị sử dụng**Trình chạy thử nghiệm Node.js**(`tests/unit/`), chạy qua `npm run test:unit`. Mã nguồn trong `src/` là**TypeScript**(`.ts`/`.tsx`); không gian làm việc `open-sse/` vẫn là JavaScript (`.js`). -8. Trang cài đặt được tổ chức thành 5 tab: Bảo mật, Định tuyến (6 chiến lược toàn cầu: điền trước, quay vòng, p2c, ngẫu nhiên, ít sử dụng nhất, tối ưu hóa chi phí), Khả năng phục hồi (giới hạn tốc độ có thể chỉnh sửa, ngắt mạch, chính sách), AI (ngân sách suy nghĩ, lời nhắc hệ thống, bộ nhớ đệm nhắc nhở), Nâng cao (proxy).## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- Build từ nguồn: `npm run build` -- Xây dựng hình ảnh Docker: `docker build -t omniroute .` -- Bắt đầu dịch vụ và xác minh: -- `NHẬN/api/cài đặt` -- `NHẬN /api/v1/model` -- URL cơ sở mục tiêu CLI phải là `http://:20128/v1` khi `PORT=20128` +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` +- `GET /api/v1/models` +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/vi/docs/FEATURES.md b/docs/i18n/vi/docs/FEATURES.md index a9fae9fb9f..4aba719ee0 100644 --- a/docs/i18n/vi/docs/FEATURES.md +++ b/docs/i18n/vi/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -Hướng dẫn trực quan cho mọi phần của bảng điều khiển OmniRoute.--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -Quản lý kết nối của nhà cung cấp AI: Nhà cung cấp OAuth (Claude Code, Codex, Gemini CLI), nhà cung cấp khóa API (Groq, DeepSeek, OpenRouter) và nhà cung cấp miễn phí (Qoder, Qwen, Kiro). Tài khoản Kiro bao gồm theo dõi số dư tín dụng — các khoản tín dụng còn lại, tổng trợ cấp và ngày gia hạn hiển thị trong Bảng điều khiển → Mức sử dụng.![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -Tạo các tổ hợp định tuyến mô hình với 6 chiến lược: ưu tiên, có trọng số, quay vòng, ngẫu nhiên, ít được sử dụng nhất và tối ưu hóa chi phí. Mỗi tổ hợp kết hợp nhiều mô hình với tính năng dự phòng tự động và bao gồm các mẫu nhanh cũng như kiểm tra mức độ sẵn sàng.![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -Phân tích sử dụng toàn diện với mức tiêu thụ mã thông báo, ước tính chi phí, bản đồ nhiệt hoạt động, biểu đồ phân phối hàng tuần và phân tích theo từng nhà cung cấp.![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -Giám sát thời gian thực: thời gian hoạt động, bộ nhớ, phiên bản, phần trăm độ trễ (p50/p95/p99), thống kê bộ đệm và trạng thái ngắt mạch của nhà cung cấp.![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -Bốn chế độ để gỡ lỗi các bản dịch API:**Playground**(trình chuyển đổi định dạng),**Chat Test**(yêu cầu trực tiếp),**Test Bench**(kiểm tra hàng loạt) và**Live Monitor**(luồng thời gian thực).![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Kiểm tra bất kỳ mô hình nào trực tiếp từ bảng điều khiển. Chọn nhà cung cấp, mô hình và điểm cuối, viết lời nhắc bằng Trình soạn thảo Monaco, truyền phát phản hồi trong thời gian thực, hủy bỏ giữa chừng và xem số liệu thời gian.--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -Chủ đề màu sắc có thể tùy chỉnh cho toàn bộ bảng điều khiển. Chọn từ 7 màu cài sẵn (San hô, Xanh lam, Đỏ, Xanh lục, Tím, Cam, Lục lam) hoặc tạo chủ đề tùy chỉnh bằng cách chọn bất kỳ màu lục giác nào. Hỗ trợ chế độ sáng, tối và hệ thống.--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -Bảng cài đặt toàn diện với các tab: +Comprehensive settings panel with tabs: --**Chung**— Lưu trữ hệ thống, quản lý sao lưu (xuất/nhập cơ sở dữ liệu) -**Giao diện**— Bộ chọn chủ đề (tối/sáng/hệ thống), cài đặt trước chủ đề màu và màu tùy chỉnh, khả năng hiển thị nhật ký tình trạng, kiểm soát khả năng hiển thị mục thanh bên -**Bảo mật**— Bảo vệ điểm cuối API, chặn nhà cung cấp tùy chỉnh, lọc IP, thông tin phiên -**Định tuyến**— Bí danh mô hình, xuống cấp tác vụ nền -**Khả năng phục hồi**— Duy trì giới hạn tỷ lệ, điều chỉnh ngắt mạch, tự động vô hiệu hóa tài khoản bị cấm, giám sát hết hạn của nhà cung cấp -**Nâng cao**— Ghi đè cấu hình, quá trình kiểm tra cấu hình, chế độ xuống cấp dự phòng![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -Cấu hình bằng một cú nhấp chuột cho các công cụ mã hóa AI: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, AntiGravity, Cline, Continue, Cursor và Factory Droid. Tính năng áp dụng/đặt lại cấu hình tự động, cấu hình kết nối và ánh xạ mô hình.![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -Bảng điều khiển để khám phá và quản lý các tác nhân CLI. Hiển thị một lưới gồm 14 tác nhân tích hợp (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) với: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**Trạng thái cài đặt**— Đã cài đặt / Không tìm thấy khi phát hiện phiên bản -**Huy hiệu giao thức**— stdio, HTTP, v.v. -**Tác nhân tùy chỉnh**— Đăng ký bất kỳ công cụ CLI nào thông qua biểu mẫu (tên, nhị phân, lệnh phiên bản, đối số sinh sản) -**Khớp dấu vân tay CLI**— Chuyển đổi theo nhà cung cấp để khớp với chữ ký yêu cầu CLI gốc, giảm rủi ro bị cấm trong khi vẫn bảo toàn IP proxy--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -Tạo hình ảnh, video và nhạc từ bảng điều khiển. Hỗ trợ OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open và MusicGen.--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -Ghi nhật ký yêu cầu theo thời gian thực với tính năng lọc theo nhà cung cấp, kiểu máy, tài khoản và khóa API. Hiển thị mã trạng thái, mức sử dụng mã thông báo, độ trễ và chi tiết phản hồi.![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -Điểm cuối API hợp nhất của bạn với phân tích chức năng: Hoàn thành cuộc trò chuyện, API phản hồi, Nhúng, Tạo hình ảnh, Xếp hạng lại, Phiên âm âm thanh, Chuyển văn bản thành giọng nói, Kiểm duyệt và khóa API đã đăng ký. Tích hợp Cloudflare Quick Tunnel và hỗ trợ proxy đám mây để truy cập từ xa.![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -Tạo, xác định phạm vi và thu hồi các khóa API. Mỗi khóa có thể được giới hạn ở những kiểu máy/nhà cung cấp cụ thể có quyền truy cập đầy đủ hoặc quyền chỉ đọc. Quản lý khóa trực quan với tính năng theo dõi việc sử dụng.--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -Theo dõi hành động quản trị bằng cách lọc theo loại hành động, tác nhân, mục tiêu, địa chỉ IP và dấu thời gian. Lịch sử sự kiện bảo mật đầy đủ.--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -Ứng dụng máy tính để bàn Electron gốc dành cho Windows, macOS và Linux. Chạy OmniRoute dưới dạng một ứng dụng độc lập có tích hợp khay hệ thống, hỗ trợ ngoại tuyến, tự động cập nhật và cài đặt chỉ bằng một cú nhấp chuột. +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -Các tính năng chính: +Key features: -- Thăm dò mức độ sẵn sàng của máy chủ (không có màn hình trống khi khởi động nguội) -- Khay hệ thống có quản lý cổng -- Chính sách bảo mật nội dung -- Khóa đơn -- Tự động cập nhật khi khởi động lại -- Giao diện người dùng có điều kiện nền tảng (đèn giao thông macOS, thanh tiêu đề mặc định của Windows/Linux) -- Đóng gói bản dựng Hardened Electron — `node_modules` được liên kết tượng trưng trong gói độc lập được phát hiện và từ chối trước khi đóng gói, ngăn chặn sự phụ thuộc thời gian chạy vào máy bản dựng (v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 Xem [`electron/README.md`](../electron/README.md) để biết tài liệu đầy đủ. +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/vi/docs/TROUBLESHOOTING.md b/docs/i18n/vi/docs/TROUBLESHOOTING.md index 66a47d77d1..db673c71a4 100644 --- a/docs/i18n/vi/docs/TROUBLESHOOTING.md +++ b/docs/i18n/vi/docs/TROUBLESHOOTING.md @@ -4,68 +4,142 @@ --- -Các vấn đề thường gặp và giải pháp cho OmniRoute.--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| Vấn đề | Giải pháp | -| ------------------------------------------ | ----------------------------------------------------------------- | --- | -| Đăng nhập lần đầu không hoạt động | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | -| Bảng điều khiển mở sai cổng | Đặt `PORT=20128` và `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| Không có nhật ký yêu cầu nào trong `logs/` | Đặt `ENABLE_REQUEST_LOGS=true` | -| EACCES: quyền bị từ chối | Đặt `DATA_DIR=/path/to/writable/dir` để ghi đè `~/.omniroute` | -| Chiến lược định tuyến không tiết kiệm | Cập nhật lên v1.4.11+ (Sửa lược đồ Zod để duy trì cài đặt) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**Nguyên nhân:**Đã hết hạn ngạch nhà cung cấp. +**Cause:** Provider quota exhausted. -**Sửa chữa:** +**Fix:** -1. Kiểm tra trình theo dõi hạn ngạch trên trang tổng quan -2. Sử dụng kết hợp với các tầng dự phòng -3. Chuyển sang cấp rẻ hơn/miễn phí### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**Lý do:**Đã hết hạn mức đăng ký. +### Rate Limiting -**Sửa chữa:** +**Cause:** Subscription quota exhausted. -- Thêm dự phòng: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- Sử dụng GLM/MiniMax làm bản sao lưu giá rẻ### OAuth Token Expired +**Fix:** -OmniRoute tự động làm mới mã thông báo. Nếu vấn đề vẫn tiếp diễn: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. Bảng điều khiển → Nhà cung cấp → Kết nối lại -2. Xóa và thêm lại kết nối nhà cung cấp--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. Xác minh `BASE_URL` trỏ đến phiên bản đang chạy của bạn (ví dụ: `http://localhost:20128`) -2. Xác minh `CLOUD_URL` trỏ đến điểm cuối đám mây của bạn (ví dụ: `https://omniroute.dev`) -3. Giữ các giá trị `NEXT_PUBLIC_*` được căn chỉnh với các giá trị phía máy chủ### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**Triệu chứng:**`Mã thông báo không mong đợi 'd'...` trên điểm cuối đám mây đối với các cuộc gọi không phát trực tuyến. +### Cloud `stream=false` Returns 500 -**Lý do:**Ngược dòng trả về tải trọng SSE trong khi khách hàng mong đợi JSON. +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**Giải pháp:**Sử dụng `stream=true` cho cuộc gọi trực tiếp qua đám mây. Thời gian chạy cục bộ bao gồm dự phòng SSE→JSON.### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. Tạo khóa mới từ bảng điều khiển cục bộ (`/api/keys`) -2. Chạy đồng bộ đám mây: Bật Đám mây → Đồng bộ hóa ngay -3. Khóa cũ/không được đồng bộ hóa vẫn có thể trả về `401` trên đám mây--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. Kiểm tra các trường thời gian chạy: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. Đối với chế độ di động: sử dụng mục tiêu hình ảnh `runner-cli` (CLS đi kèm) -3. Đối với chế độ gắn máy chủ: đặt `CLI_EXTRA_PATHS` và gắn thư mục bin máy chủ ở chế độ chỉ đọc -4. Nếu `installed=true` và `runnable=false`: đã tìm thấy nhị phân nhưng kiểm tra tình trạng không thành công### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. Kiểm tra số liệu thống kê sử dụng trong Bảng điều khiển → Mức sử dụng -2. Chuyển model chính sang GLM/MiniMax -3. Sử dụng bậc miễn phí (Gemini CLI, Qoder) cho các tác vụ không quan trọng -4. Đặt ngân sách chi phí cho mỗi khóa API: Bảng điều khiển → Khóa API → Ngân sách--- +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax +3. Use free tier (Gemini CLI, Qoder) for non-critical tasks +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -Đặt `ENABLE_REQUEST_LOGS=true` trong tệp `.env` của bạn. Nhật ký xuất hiện trong thư mục `logs/`.### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- Trạng thái chính: `${DATA_DIR}/storage.sqlite` (nhà cung cấp, tổ hợp, bí danh, khóa, cài đặt) -- Cách sử dụng: Các bảng SQLite trong `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + tùy chọn `${DATA_DIR}/log.txt` và `${DATA_DIR}/call_logs/` -- Nhật ký yêu cầu: `/logs/...` (khi `ENABLE_REQUEST_LOGS=true`)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -Khi cầu dao của nhà cung cấp MỞ, các yêu cầu sẽ bị chặn cho đến khi hết thời gian hồi chiêu. +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**Sửa chữa:** +**Fix:** -1. Đi tới**Bảng điều khiển → Cài đặt → Khả năng phục hồi** -2. Kiểm tra thẻ cầu dao của nhà cung cấp bị ảnh hưởng -3. Nhấp vào**Đặt lại tất cả**để xóa tất cả các bộ ngắt hoặc đợi hết thời gian hồi chiêu -4. Xác minh nhà cung cấp thực sự có sẵn trước khi đặt lại### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -Nếu nhà cung cấp liên tục chuyển sang trạng thái MỞ: +### Provider keeps tripping the circuit breaker -1. Kiểm tra**Bảng điều khiển → Sức khỏe → Tình trạng nhà cung cấp**để biết kiểu lỗi -2. Đi tới**Cài đặt → Khả năng phục hồi → Hồ sơ nhà cung cấp**và tăng ngưỡng thất bại -3. Kiểm tra xem nhà cung cấp có thay đổi giới hạn API hay yêu cầu xác thực lại không -4. Xem lại phép đo từ xa về độ trễ - độ trễ cao có thể gây ra lỗi dựa trên thời gian chờ--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- Đảm bảo bạn đang sử dụng đúng tiền tố: `deepgram/nova-3` hoặc `assemblyai/best` -- Xác minh nhà cung cấp được kết nối trong**Bảng điều khiển → Nhà cung cấp**### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- Kiểm tra các định dạng âm thanh được hỗ trợ: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` -- Xác minh kích thước tệp nằm trong giới hạn của nhà cung cấp (thường < 25 MB) -- Kiểm tra tính hợp lệ của khóa API nhà cung cấp trong thẻ nhà cung cấp--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -Sử dụng**Trang tổng quan → Trình dịch**để gỡ lỗi các vấn đề dịch định dạng: +Use **Dashboard → Translator** to debug format translation issues: -| Chế độ | Khi nào nên sử dụng | -| ----------------------------- | ------------------------------------------------------------------------------------------------------------ | ------------------------ | -| **Sân chơi** | So sánh các định dạng đầu vào/đầu ra cạnh nhau — dán một yêu cầu không thành công để xem nó dịch như thế nào | -| **Người kiểm tra trò chuyện** | Gửi tin nhắn trực tiếp và kiểm tra toàn bộ tải trọng yêu cầu/phản hồi bao gồm các tiêu đề | -| **Bàn thử nghiệm** | Chạy thử nghiệm hàng loạt trên các kết hợp định dạng để tìm ra bản dịch nào bị lỗi | -| **Màn hình trực tiếp** | Xem luồng yêu cầu theo thời gian thực để nắm bắt các vấn đề dịch thuật không liên tục | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**Thẻ tư duy không xuất hiện**— Kiểm tra xem nhà cung cấp mục tiêu có hỗ trợ tư duy và cài đặt ngân sách tư duy hay không -**Giảm cuộc gọi công cụ**— Một số bản dịch định dạng có thể loại bỏ các trường không được hỗ trợ; xác minh ở chế độ Playground -**Thiếu lời nhắc hệ thống**— Claude và Gemini xử lý lời nhắc hệ thống theo cách khác nhau; kiểm tra đầu ra bản dịch -**SDK trả về chuỗi thô thay vì đối tượng**— Đã sửa trong v1.1.0: trình khử trùng phản hồi hiện loại bỏ các trường không chuẩn (`x_groq`, `usage_breakdown`, v.v.) gây ra lỗi xác thực OpenAI SDK Pydantic -**GLM/ERNIE từ chối vai trò `system`**— Đã sửa trong v1.1.0: bộ chuẩn hóa vai trò tự động hợp nhất các thông báo hệ thống thành thông báo người dùng cho các kiểu máy không tương thích -**`vai trò nhà phát triển` không được nhận dạng**— Đã sửa trong v1.1.0: tự động chuyển đổi thành `system` cho các nhà cung cấp không phải OpenAI -**`json_schema` không hoạt động với Gemini**— Đã sửa trong v1.1.0: `response_format` hiện được chuyển đổi thành `responseMimeType` + `responseSchema` của Gemini--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- Giới hạn tỷ lệ tự động chỉ áp dụng cho nhà cung cấp khóa API (không phải OAuth/đăng ký) -- Xác minh**Cài đặt → Khả năng phục hồi → Hồ sơ nhà cung cấp**đã bật giới hạn tỷ lệ tự động -- Kiểm tra xem nhà cung cấp có trả về mã trạng thái `429` hoặc tiêu đề `Thử lại sau` không### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -Hồ sơ nhà cung cấp hỗ trợ các cài đặt này: +### Tuning exponential backoff --**Độ trễ cơ bản**— Thời gian chờ ban đầu sau lần thất bại đầu tiên (mặc định: 1 giây) -**Độ trễ tối đa**— Giới hạn thời gian chờ tối đa (mặc định: 30 giây) -**Hệ số**— Độ trễ tăng lên bao nhiêu cho mỗi lần thất bại liên tiếp (mặc định: 2x)### Anti-thundering herd +Provider profiles support these settings: -Khi nhiều yêu cầu đồng thời gặp phải một nhà cung cấp có tốc độ giới hạn, OmniRoute sử dụng mutex + giới hạn tốc độ tự động để tuần tự hóa các yêu cầu và ngăn chặn lỗi xếp tầng. Điều này là tự động đối với các nhà cung cấp khóa API.--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -Một số người dùng OmniRoute đặt cổng phía trước RAG hoặc ngăn tác nhân. Trong các thiết lập đó, người ta thường thấy một mẫu lạ: OmniRoute có vẻ ổn (nhà cung cấp hoạt động, cấu hình định tuyến ổn, không có cảnh báo giới hạn tốc độ) nhưng câu trả lời cuối cùng vẫn sai. +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -Trong thực tế, những sự cố này thường đến từ đường ống RAG xuôi dòng chứ không phải từ chính cổng. +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -Nếu bạn muốn có một từ vựng chung để mô tả những lỗi đó, bạn có thể sử dụng Bản đồ vấn đề WFGY, một tài nguyên văn bản giấy phép MIT bên ngoài xác định mười sáu mẫu lỗi RAG / LLM định kỳ. Ở mức độ cao, nó bao gồm: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- thu hồi trôi dạt và ranh giới bối cảnh bị phá vỡ -- các chỉ mục và cửa hàng vector trống hoặc cũ -- nhúng và không khớp ngữ nghĩa -- các vấn đề về lắp ráp và ngữ cảnh nhanh chóng -- suy sụp logic và câu trả lời quá tự tin -- thất bại phối hợp chuỗi dài và đại lý -- bộ nhớ đa tác nhân và trôi dạt vai trò -- vấn đề về triển khai và đặt hàng bootstrap +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -Ý tưởng rất đơn giản: +The idea is simple: -1. Khi bạn điều tra một phản hồi không tốt, hãy nắm bắt: - - nhiệm vụ và yêu cầu của người dùng - - kết hợp tuyến đường hoặc nhà cung cấp trong OmniRoute - - bất kỳ bối cảnh RAG nào được sử dụng ở phía dưới (tài liệu được truy xuất, lệnh gọi công cụ, v.v.) -2. Ánh xạ sự cố tới một hoặc hai số Bản đồ vấn đề WFGY (`No.1` … `No.16`). -3. Lưu số này vào bảng điều khiển, sổ ghi chép hoặc trình theo dõi sự cố của riêng bạn bên cạnh nhật ký OmniRoute. -4. Sử dụng trang WFGY tương ứng để quyết định xem bạn có cần thay đổi chiến lược ngăn xếp, truy xuất hoặc định tuyến RAG của mình hay không. +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -Toàn văn và công thức nấu ăn cụ thể có tại đây (giấy phép MIT, chỉ văn bản): +Full text and concrete recipes live here (MIT license, text only): -[ĐỌC Bản đồ vấn đề WFGY](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) +[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -Bạn có thể bỏ qua phần này nếu bạn không chạy RAG hoặc đường dẫn tác nhân phía sau OmniRoute.--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**Vấn đề về GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Kiến trúc**: Xem [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) để biết chi tiết nội bộ -**Tham khảo API**: Xem [`docs/API_REFERENCE.md`](API_REFERENCE.md) để biết tất cả các điểm cuối -**Bảng điều khiển sức khỏe**: Kiểm tra**Bảng điều khiển → Sức khỏe**để biết trạng thái hệ thống theo thời gian thực -**Trình dịch**: Sử dụng**Bảng điều khiển → Trình dịch**để gỡ lỗi các vấn đề về định dạng +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/vi/llm.txt b/docs/i18n/vi/llm.txt new file mode 100644 index 0000000000..767b313110 --- /dev/null +++ b/docs/i18n/vi/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (Tiếng Việt) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## Tổng quan + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### Bảo mật +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/docs/i18n/zh-CN/README.md b/docs/i18n/zh-CN/README.md index 7741b2c59e..2fa06c9ccf 100644 --- a/docs/i18n/zh-CN/README.md +++ b/docs/i18n/zh-CN/README.md @@ -4,11 +4,14 @@ --- + ### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback. -_您的通用 API 代理 — 一个端点、60 多个提供商、零停机时间。现在拥有**MCP 服务器(25 个工具)**、**A2A 协议**、**内存/技能系统**和**Electron 桌面应用程序**。_ +_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._ -**聊天完成 • 嵌入 • 图像生成 • 视频 • 音乐 • 音频 • 重新排名 •**网页搜索**• MCP 服务器 • A2A 协议 • 100% TypeScript**--- +**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript** + +---
@@ -39,9 +42,13 @@ _您的通用 API 代理 — 一个端点、60 多个提供商、零停机时间 [![Website](https://img.shields.io/badge/Website-omniroute.online-blue?logo=google-chrome&logoColor=white)](https://omniroute.online) [![WhatsApp](https://img.shields.io/badge/WhatsApp-Community-25D366?logo=whatsapp&logoColor=white)](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -[🌐 网站](https://omniroute.online) • [🚀 快速入门](#-quick-start) • [💡 功能](#-key-features) • [📖 文档](#-documentation) • [💰 定价](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -🌐**可用版本:**🇺🇸 [英语](README.md) | 🇧🇷 [葡萄牙语(巴西)](docs/i18n/pt-BR/README.md) | 🇪🇸 [西班牙语](docs/i18n/es/README.md) | 🇫🇷 [法语](docs/i18n/fr/README.md) | 🇮🇹 [意大利语](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [德语](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [更新](docs/i18n/ar/README.md) | 🇯🇵 [日本语](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [丹麦](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברйת](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [印尼语](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [马来语](docs/i18n/ms/README.md) | 🇳🇱 [荷兰](docs/i18n/nl/README.md) | 🇳🇴 [挪威](docs/i18n/no/README.md) | 🇵🇹 [葡萄牙语(葡萄牙)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [斯洛文尼亚](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [菲律宾语](docs/i18n/phi/README.md) | 🇨🇿 [捷克](docs/i18n/cs/README.md)--- +
+ +🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md) + +--- ## 🖼️ Main Dashboard @@ -53,554 +60,629 @@ _您的通用 API 代理 — 一个端点、60 多个提供商、零停机时间 ## 📸 Dashboard Preview -<详情> +
+Click to see dashboard screenshots -点击查看仪表板屏幕截图 +| Page | Screenshot | +| -------------- | ------------------------------------------------- | +| **Providers** | ![Providers](docs/screenshots/01-providers.png) | +| **Combos** | ![Combos](docs/screenshots/02-combos.png) | +| **Analytics** | ![Analytics](docs/screenshots/03-analytics.png) | +| **Health** | ![Health](docs/screenshots/04-health.png) | +| **Translator** | ![Translator](docs/screenshots/05-translator.png) | +| **Settings** | ![Settings](docs/screenshots/06-settings.png) | +| **CLI Tools** | ![CLI Tools](docs/screenshots/07-cli-tools.png) | +| **Usage Logs** | ![Usage](docs/screenshots/08-usage.png) | +| **Endpoints** | ![Endpoints](docs/screenshots/09-endpoint.png) | -| 页 | 截图 | -| ------------ | ---------------------------------------------- | ---------- | -| **提供商** | ![提供商](docs/screenshots/01-providers.png) | -| **组合** | ![组合](docs/screenshots/02-combos.png) | -| **分析** | ![分析](docs/screenshots/03-analytics.png) | -| **健康** | ![健康](docs/screenshots/04-health.png) | -| **翻译** | ![译者](docs/screenshots/05-translator.png) | -| **设置** | ![设置](docs/screenshots/06-settings.png) | -| **CLI 工具** | ![CLI 工具](docs/screenshots/07-cli-tools.png) | -| **使用日志** | ![用法](docs/screenshots/08-usage.png) | -| **端点** | ![端点](docs/screenshots/09-endpoint.png) |
| + --- ### 🤖 Free AI Provider for your favorite coding agents -_通过 OmniRoute 连接任何人工智能驱动的 IDE 或 CLI 工具 — 免费 API 网关,可进行无限编码。_ +_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._ -<表> - - - -OpenClaw
-张开爪 -

-<子>⭐205K - - - -NanoBot
-纳米机器人 -

-<子>⭐20.9K - - - -PicoClaw
-微微爪 -

-<子>⭐ 14.6K - - - -ZeroClaw
-零爪 -

-<子>⭐9.9K - - - -IronClaw
-铁爪 -

-<子>⭐2.1K - - - - - -OpenCode
-开放代码 -

-<子>⭐106K - - - -Codex CLI
-法典 CLI -

-<子>⭐ 60.8K - - - -克劳德代码
-克劳德代码 -

-<子>⭐ 67.3K - - - -Gemini CLI
-双子座 CLI -

-<子>⭐94.7K - - - -基洛代码
-基洛代码 -

-<子>⭐ 15.5K - - - + + + + + + + + + + + + + + + +
+ + OpenClaw
+ OpenClaw +

+ ⭐ 205K +
+ + NanoBot
+ NanoBot +

+ ⭐ 20.9K +
+ + PicoClaw
+ PicoClaw +

+ ⭐ 14.6K +
+ + ZeroClaw
+ ZeroClaw +

+ ⭐ 9.9K +
+ + IronClaw
+ IronClaw +

+ ⭐ 2.1K +
+ + OpenCode
+ OpenCode +

+ ⭐ 106K +
+ + Codex CLI
+ Codex CLI +

+ ⭐ 60.8K +
+ + Claude Code
+ Claude Code +

+ ⭐ 67.3K +
+ + Gemini CLI
+ Gemini CLI +

+ ⭐ 94.7K +
+ + Kilo Code
+ Kilo Code +

+ ⭐ 15.5K +
-📡 所有代理均通过 http://localhost:20128/v1http://cloud.omniroute.online/v1 连接 - 一个配置,无限型号和配额--- +📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota + +--- ## 🤔 Why OmniRoute? -**停止浪费金钱和达到极限:** +**Stop wasting money and hitting limits:** -- 订阅配额每月到期未使用 -- 速率限制阻止您进行中间编码 -- 昂贵的 API(每个提供商每月 20-50 美元) -- 在提供商之间手动切换 +- Subscription quota expires unused every month +- Rate limits stop you mid-coding +- Expensive APIs ($20-50/month per provider) +- Manual switching between providers -**OmniRoute 解决了这个问题:** +**OmniRoute solves this:** -- ✅**最大化订阅**- 跟踪配额,在重置之前使用所有位 -- ✅**自动回退**- 订阅 → API 密钥 → 便宜 → 免费,零停机时间 -- ✅**多帐户**- 每个提供商的帐户之间循环 -- ✅**通用**- 可与 Claude Code、Codex、Gemini CLI、Cursor、Cline、OpenClaw、任何 CLI 工具配合使用--- +- ✅ **Maximize subscriptions** - Track quota, use every bit before reset +- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime +- ✅ **Multi-account** - Round-robin between accounts per provider +- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool + +--- ## 📧 Support -> 💬**加入我们的社区!**[WhatsApp 群组](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — 获取帮助、分享提示并保持更新。 +> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated. --**网站**:[omniroute.online](https://omniroute.online) -**GitHub**:[github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**问题**:[github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**:[社区组](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**贡献**:请参阅 [CONTRIBUTING.md](CONTRIBUTING.md),打开 PR,或选择“好第一期” -**原始项目**:[9router by decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug? +- **Website**: [omniroute.online](https://omniroute.online) +- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) +- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) +- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue` +- **Original Project**: [9router by decolua](https://github.com/decolua/9router) -打开问题时,请运行 system-info 命令并附加生成的文件:```bash +### 🐛 Reporting a Bug? + +When opening an issue, please run the system-info command and attach the generated file: + +```bash npm run system-info - ``` -这会生成一个“system-info.txt”,其中包含您的 Node.js 版本、OmniRoute 版本、操作系统详细信息、已安装的 CLI 工具(qoder、gemini、claude、codex、antigravity、droid 等)、Docker/PM2 状态和系统包 - 我们快速重现您的问题所需的一切。将该文件直接附加到您的 GitHub 问题。--- +This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue. + +--- ## 🔄 How It Works ``` - ┌─────────────┐ -│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) -│ Tool │ +│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...) +│ Tool │ └──────┬──────┘ -│ http://localhost:20128/v1 -↓ + │ http://localhost:20128/v1 + ↓ ┌─────────────────────────────────────────┐ -│ OmniRoute (Smart Router) │ -│ • Format translation (OpenAI ↔ Claude) │ -│ • Quota tracking + Embeddings + Images │ -│ • Auto token refresh │ +│ OmniRoute (Smart Router) │ +│ • Format translation (OpenAI ↔ Claude) │ +│ • Quota tracking + Embeddings + Images │ +│ • Auto token refresh │ └──────┬──────────────────────────────────┘ -│ -├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI -│ ↓ quota exhausted -├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. -│ ↓ budget limit -├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) -│ ↓ budget limit -└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) + │ + ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI + │ ↓ quota exhausted + ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc. + │ ↓ budget limit + ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M) + │ ↓ budget limit + └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited) Result: Never stop coding, minimal cost - -```` +``` --- ## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases ->**每个使用 AI 工具的开发人员每天都会面临这些问题。**OmniRoute 的构建是为了解决所有这些问题 - 从成本超支到区域封锁,从损坏的 OAuth 流程到协议操作和企业可观察性。 +> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability. -<详情> -💸 1.“我支付了昂贵的订阅费用,但仍然受到限制的干扰” +
+💸 1. "I pay for an expensive subscription but still get interrupted by limits" -开发人员每月为 Claude Pro、Codex Pro 或 GitHub Copilot 支付 20-200 美元。即使付费,配额也有上限——5 小时的使用时间、每周限制或每分钟的费率限制。在编码会话中,提供商停止响应,开发人员失去流量和生产力。 +Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity. -**OmniRoute 如何解决:** +**How OmniRoute solves it:** --**智能 4 层回退**— 如果订阅配额用完,自动重定向到 API 密钥 → 便宜 → 免费,零手动干预 --**提供商限制跟踪**— 按服务器端计划刷新缓存配额快照(默认“PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70”),并在 UI 中提供手动刷新 --**多帐户支持**— 每个提供商有多个帐户,具有自动循环 — 当一个帐户用完时,切换到下一个帐户 --**自定义组合**— 可定制的后备链,具有 9 种平衡策略(优先级、加权、先填充、循环、P2C、随机、最少使用、成本优化、严格随机) --**Codex Business Quotas**— 直接在仪表板中监控业务/团队工作空间配额
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention +- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI +- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next +- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**) +- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard -<详情> -🔌 2.“我需要使用多个提供程序,但每个提供程序都有不同的 API” + -OpenAI 使用一种格式,Claude(Anthropic)使用另一种格式,Gemini 使用另一种格式。如果开发人员想要测试来自不同提供商的模型或在它们之间进行回退,他们需要重新配置 SDK、更改端点、处理不兼容的格式。自定义提供程序(FriendLI、NIM)具有非标准模型端点。 +
+🔌 2. "I need to use multiple providers but each has a different API" -**OmniRoute 如何解决:** +OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints. --**统一端点**— 单个“http://localhost:20128/v1”充当所有 60 多个提供商的代理 --**格式翻译**— 自动且透明:OpenAI ↔ Claude ↔ Gemini ↔ Responses API --**响应清理**— 删除破坏 OpenAI SDK v1.83+ 的非标准字段(`x_groq`、`usage_breakdown`、`service_tier`) --**角色标准化**— 对于非 OpenAI 提供商,将“开发人员”转换为“系统”; GLM/ERNIE 的“系统”→“用户” --**思考标签提取**— 将 DeepSeek R1 等模型中的“”块提取为标准化的“reasoning_content” --**Gemini 的结构化输出**— `json_schema` → `responseMimeType`/`responseSchema` 自动转换 --**`stream` 默认为 `false`**— 与 OpenAI 规范保持一致,避免 Python/Rust/Go SDK 中出现意外的 SSE
+**How OmniRoute solves it:** -<详情> -🌐 3.“我的人工智能提供商屏蔽了我的地区/国家” +- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers +- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API +- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+ +- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE +- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content` +- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion +- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs -OpenAI/Codex 等提供商会阻止来自某些地理区域的访问。用户在 OAuth 和 API 连接期间会收到诸如“unsupported_country_region_territory”之类的错误。这对于发展中国家的开发商来说尤其令人沮丧。 + -**OmniRoute 如何解决:** +
+🌐 3. "My AI provider blocks my region/country" --**3 级代理配置**— 3 级可配置代理:全局(所有流量)、每个提供商(仅一个提供商)和每个连接/密钥 --**颜色编码的代理徽章**— 视觉指示器:🟢 全局代理、🟡 提供商代理、🔵 连接代理,始终显示 IP --**通过代理进行 OAuth 令牌交换**— OAuth 流程也通过代理,解决了“unsupported_country_region_territory”问题 --**通过代理进行连接测试**— 连接测试使用配置的代理(不再直接绕过) --**SOCKS5 支持**— 对出站路由的完整 SOCKS5 代理支持 --**TLS 指纹欺骗**- 通过“wreq-js”的类似浏览器的 TLS 指纹来绕过机器人检测 --**🔏 CLI 指纹匹配**— 重新排序标头和正文字段以匹配本机 CLI 二进制签名,从而大大降低帐户标记风险。代理 IP 被保留 — 您同时获得隐秘**和**IP 屏蔽
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries. -<详情> -🆓 4.“我想用AI来编码,但我没有钱” +**How OmniRoute solves it:** -并不是每个人都能每月支付 20-200 美元来订阅 AI。来自新兴国家的学生、开发人员、业余爱好者和自由职业者需要以零成本获得优质模型。 +- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key +- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP +- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory` +- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass) +- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection +- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously -**OmniRoute 如何解决:** + --**内置免费层提供商**- 对 100% 免费提供商的本机支持:Qoder(通过 OAuth 实现 5 个无限模型:kimi-k2-thinking、qwen3-coder-plus、deepseek-r1、minimax-m2、kimi-k2)、Qwen(4 个无限模型:qwen3-coder-plus、qwen3-coder-flash、qwen3-coder-next、vision-model)、Kiro(Claude + AWS Builder ID 免费)、Gemini CLI(180K 代币/月免费) --**Ollama Cloud**— 位于 `api.ollama.com` 的云托管 Ollama 模型,具有免费的“轻度使用”级别;使用 `ollamacloud/` 前缀 --**仅限免费组合**— 链 `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = 0 美元/月,零停机时间 --**NVIDIA NIM 免费访问**— ~40 RPM 开发者永远免费访问 build.nvidia.com 上的 70 多个模型(从积分过渡到纯粹的速率限制) --**成本优化策略**— 自动选择最便宜的可用提供商的路由策略 +
+🆓 4. "I want to use AI for coding but I have no money" -<详情> -🔒 5.“我需要保护我的 AI 网关免遭未经授权的访问” +Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost. -当将人工智能网关暴露到网络(LAN、VPS、Docker)时,任何拥有该地址的人都可以消耗开发者的代币/配额。如果没有保护,API 很容易被误用、提示注入和滥用。 +**How OmniRoute solves it:** -**OmniRoute 如何解决:** +- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free) +- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix +- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime +- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits) +- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider --**API 密钥管理**— 使用专用的“/dashboard/api-manager”页面生成、轮换和确定每个提供商的范围 --**模型级权限**- 将 API 密钥限制为特定模型(`openai/*`、通配符模式),并具有“允许全部”/“限制”切换功能 --**API 端点保护**— 需要“/v1/models”密钥并阻止列表中的特定提供商 --**Auth Guard + CSRF 保护**- 所有仪表板路由均受“withAuth”中间件 + CSRF 令牌保护 --**速率限制器**— 通过可配置窗口限制每个 IP 的速率 --**IP 过滤**— 用于访问控制的允许列表/阻止列表 --**Prompt Injection Guard**— 针对恶意提示模式的清理 --**AES-256-GCM 加密**— 静态加密的凭证
+ -<详情> -🛑 6.“我的提供商宕机了,我失去了编码流程” +
+🔒 5. "I need to protect my AI gateway from unauthorized access" -AI 提供商可能会变得不稳定、返回 5xx 错误或达到临时速率限制。如果开发人员依赖于单一提供商,他们就会受到干扰。如果没有断路器,重复重试可能会使应用程序崩溃。 +When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse. -**OmniRoute 如何解决:** +**How OmniRoute solves it:** --**每个模型的断路器**- 自动打开/关闭,具有可配置的阈值和冷却时间(关闭/打开/半打开),每个模型的范围以避免级联块 --**指数退避**— 渐进式重试延迟 --**Anti-Thundering Herd**— 互斥锁 + 信号量保护,防止并发重试风暴 --**组合后备链**— 如果主要提供商发生故障,则自动从该链中掉下来,无需干预 --**组合断路器**— 自动禁用组合链中出现故障的提供商 --**运行状况仪表板**— 正常运行时间监控、断路器状态、锁定、缓存统计、p50/p95/p99 延迟
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page +- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle +- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing +- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens +- **Rate Limiter** — Per-IP rate limiting with configurable windows +- **IP Filtering** — Allowlist/blocklist for access control +- **Prompt Injection Guard** — Sanitization against malicious prompt patterns +- **AES-256-GCM Encryption** — Credentials encrypted at rest -<详情> -🔧 7.“配置每个AI工具都是繁琐且重复的” + -开发人员使用 Cursor、Claude Code、Codex CLI、OpenClaw、Gemini CLI、Kilo Code...每个工具都需要不同的配置(API 端点、密钥、模型)。切换提供商或模型时重新配置是浪费时间。 +
+🛑 6. "My provider went down and I lost my coding flow" -**OmniRoute 如何解决:** +AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application. --**CLI 工具仪表板**— 专用页面,可一键设置 Claude Code、Codex CLI、OpenClaw、Kilo Code、Antigravity、Cline --**GitHub Copilot 配置生成器**— 通过批量模型选择为 VS Code 生成 `chatLanguageModels.json` --**入门向导**— 为首次使用的用户提供 4 步设置指导 --**一个端点,所有型号**— 配置 `http://localhost:20128/v1` 一次,访问 60 多个提供商
+**How OmniRoute solves it:** -<详情> -🔑 8.“管理来自多个提供商的 OAuth 令牌简直就是地狱” +- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks +- **Exponential Backoff** — Progressive retry delays +- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms +- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention +- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain +- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency -Claude Code、Codex、Gemini CLI、Copilot — 全部使用带有过期令牌的 OAuth 2.0。开发人员需要不断地重新进行身份验证,处理“client_secret is Missing”、“redirect_uri_mismatch”以及远程服务器上的故障。 LAN/VPS 上的 OAuth 问题尤其严重。 + -**OmniRoute 如何解决:** +
+🔧 7. "Configuring each AI tool is tedious and repetitive" --**自动令牌刷新**— OAuth 令牌在过期前在后台刷新 --**OAuth 2.0 (PKCE) 内置**— Claude Code、Codex、Gemini CLI、Copilot、Kiro、Qwen、Qoder 的自动流程 --**多帐户 OAuth**— 每个提供商通过 JWT/ID 令牌提取多个帐户 --**OAuth LAN/远程修复**— `redirect_uri` 的私有 IP 检测 + 远程服务器的手动 URL 模式 --**Nginx 背后的 OAuth**— 使用 `window.location.origin` 实现反向代理兼容性 --**远程 OAuth 指南**— VPS/Docker 上的 Google Cloud 凭据分步指南
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time. -<详情> -📊 9.“我不知道我花了多少钱或在哪里” +**How OmniRoute solves it:** -开发商使用多个付费提供商,但对支出没有统一的看法。每个提供商都有自己的计费仪表板,但没有统一的视图。意外的成本可能会不断增加。 +- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline +- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection +- **Onboarding Wizard** — Guided 4-step setup for first-time users +- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers -**OmniRoute 如何解决:** + --**成本分析仪表板**— 每个提供商的每个代币成本跟踪和预算管理 --**每层预算限制**— 触发自动回退的每层支出上限 --**按型号定价配置**— 每个型号的可配置价格 --**每个 API 密钥的使用统计信息**— 每个密钥的请求计数和上次使用的时间戳 --**分析仪表板**— 统计卡、模型使用图表、包含成功率和延迟的提供商表 +
+🔑 8. "Managing OAuth tokens from multiple providers is hell" -<详情> -🐛 10.“我无法诊断 AI 调用中的错误和问题” +Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic. -当调用失败时,开发人员不知道这是否是速率限制、令牌过期、格式错误或提供商错误。跨不同终端的碎片日志。如果没有可观察性,调试就是反复试验。 +**How OmniRoute solves it:** -**OmniRoute 如何解决:** +- **Auto Token Refresh** — OAuth tokens refresh in background before expiration +- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder +- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction +- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers +- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility +- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker --**统一日志仪表板**— 4 个选项卡:请求日志、代理日志、审核日志、控制台 --**控制台日志查看器**— 实时终端式查看器,具有颜色编码级别、自动滚动、搜索、过滤功能 --**SQLite 代理日志**— 服务器重新启动后仍保留的持久日志 --**Translator Playground**— 4 种调试模式:Playground(格式翻译)、Chat Tester(往返)、Test Bench(批量)、Live Monitor(实时) --**请求遥测**— p50/p95/p99 延迟 + X-Request-Id 跟踪 --**基于文件的日志记录与轮换**— 应用程序日志按大小、保留天数和存档计数轮换;通话记录工件按保留天数和文件计数轮换 --**系统信息报告**— `npm run system-info` 生成包含完整环境(Node 版本、OmniRoute 版本、操作系统、CLI 工具、Docker/PM2 状态)的 `system-info.txt`。报告问题时附上它以进行即时分类。
+ -<详情> -🏗️ 11.“部署和维护网关很复杂” +
+📊 9. "I don't know how much I'm spending or where" -跨不同环境(本地、VPS、Docker、云)安装、配置和维护 AI 代理是一项劳动密集型工作。硬编码路径、目录上的“EACCES”、端口冲突和跨平台构建等问题会增加摩擦。 +Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up. -**OmniRoute 如何解决:** +**How OmniRoute solves it:** --**npm 全局安装**— `npm install -gomniroute &&omniroute` — 完成 --**Docker 多平台**— AMD64 + ARM64 本机(Apple Silicon、AWS Graviton、Raspberry Pi) --**Docker Compose Profiles**— `base`(无 CLI 工具)和 `cli`(使用 Claude Code、Codex、OpenClaw) --**Electron 桌面应用程序**— 适用于 Windows/macOS/Linux 的本机应用程序,带系统托盘、自动启动、离线模式 --**分割端口模式**— API 和仪表板位于单独的端口上,适用于高级场景(反向代理、容器网络) --**云同步**— 通过 Cloudflare Workers 跨设备配置同步 --**数据库备份**— 自动备份、恢复、导出和导入所有设置,使用“DISABLE_SQLITE_AUTO_BACKUP”进行外部管理备份
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider +- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback +- **Per-Model Pricing Configuration** — Configurable prices per model +- **Usage Statistics Per API Key** — Request count and last-used timestamp per key +- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency -<详情> -🌍 12.“界面只有英文,我的团队不会说英语” + -非英语国家的团队,尤其是拉丁美洲、亚洲和欧洲的团队,在纯英文界面上遇到了困难。语言障碍会降低采用率并增加配置错误。 +
+🐛 10. "I can't diagnose errors and problems in AI calls" -**OmniRoute 如何解决:** +When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error. --**仪表板 i18n — 30 种语言**— 所有 500 多个按键已翻译,包括阿拉伯语、保加利亚语、丹麦语、德语、西班牙语、芬兰语、法语、希伯来语、印地语、匈牙利语、印度尼西亚语、意大利语、日语、韩语、马来语、荷兰语、挪威语、波兰语、葡萄牙语(PT/BR)、罗马尼亚语、俄语、斯洛伐克语、瑞典语、泰语、乌克兰语、越南语、中文、菲律宾语、英语 --**RTL 支持**— 从右到左支持阿拉伯语和希伯来语 --**多语言自述文件**— 30 个完整的文档翻译 --**语言选择器**— 标题中的地球图标用于实时切换
+**How OmniRoute solves it:** -<详情> -🔄 13.“我需要的不仅仅是聊天 - 我需要嵌入、图像、音频” +- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console +- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter +- **SQLite Proxy Logs** — Persistent logs that survive server restarts +- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time) +- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing +- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count +- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage. -人工智能不仅仅是完成聊天。开发人员需要生成图像、转录音频、为 RAG 创建嵌入、重新排列文档以及审核内容。每个 API 都有不同的端点和格式。 + -**OmniRoute 如何解决:** +
+🏗️ 11. "Deploying and maintaining the gateway is complex" --**Embeddings**— `/v1/embeddings` 具有 6 个提供商和 9 个以上模型 --**图像生成**— `/v1/images/Generations` 拥有 10 个提供商和 20 多个模型(OpenAI、xAI、Together、Fireworks、Nebius、Hyperbolic、NanoBanana、Antigravity、SD WebUI、ComfyUI) --**文本到视频**— `/v1/videos/ Generations` — ComfyUI(AnimateDiff、SVD)和 SD WebUI --**文本到音乐**— `/v1/music/generations` — ComfyUI(稳定音频开放,MusicGen) --**音频转录**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM、HuggingFace、Qwen3 --**文本转语音**— `/v1/audio/speech` — ElevenLabs、Nvidia NIM、HuggingFace、Coqui、Tortoise、Qwen3、**Inworld**、**Cartesia**、**PlayHT**、+ 现有提供商 --**审核**— `/v1/moderations` — 内容安全检查 --**重新排名**— `/v1/rerank` — 文档相关性重新排名 --**响应 API**— 对 Codex 的完整 `/v1/responses` 支持
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction. -<详情> -🧪 14.“我无法测试和比较不同模型的质量” +**How OmniRoute solves it:** -开发人员想知道哪种模型最适合他们的用例(代码、翻译、推理),但手动比较速度很慢。不存在集成的评估工具。 +- **npm global install** — `npm install -g omniroute && omniroute` — done +- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi) +- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw) +- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode +- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking) +- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers +- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups -**OmniRoute 如何解决:** + --**LLM 评估**— 黄金套装测试,包含 10 个预加载案例,涵盖问候语、数学、地理、代码生成、JSON 合规性、翻译、降价、安全拒绝 --**4 种匹配策略**— `exact`、`contains`、`regex`、`custom` (JS 函数) --**Translator Playground 测试台**— 使用多个输入和预期输出进行批量测试、跨提供商比较 --**聊天测试器**— 带有视觉响应渲染的完整往返 --**实时监控**— 流经代理的所有请求的实时流 +
+🌍 12. "The interface is English-only and my team doesn't speak English" -<详情> -📈 15.“我需要在不损失性能的情况下进行扩展” +Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors. -随着请求量的增长,如果不缓存相同的问题,就会产生重复的成本。如果没有幂等性,重复的请求就会浪费处理。必须遵守每个提供商的速率限制。 +**How OmniRoute solves it:** -**OmniRoute 如何解决:** +- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English +- **RTL Support** — Right-to-left support for Arabic and Hebrew +- **Multi-Language READMEs** — 30 complete documentation translations +- **Language Selector** — Globe icon in header for real-time switching --**语义缓存**- 两层缓存(签名+语义)降低成本和延迟 --**请求幂等性**— 相同请求的 5 秒重复数据删除窗口 --**速率限制检测**— 每个提供商的 RPM、最小间隙和最大并发跟踪 --**可编辑的速率限制**— 可在“设置”→“持久弹性”中配置默认值 --**API 密钥验证缓存**— 用于提高生产性能的 3 层缓存 --**带有遥测功能的运行状况仪表板**— p50/p95/p99 延迟、缓存统计数据、正常运行时间
+ -<详情> -🤖 16.“我想全局控制模型行为” +
+🔄 13. "I need more than chat — I need embeddings, images, audio" -希望所有响应都以特定语言、特定语气或想要限制推理标记的开发人员。在每个工具/请求中配置此功能是不切实际的。 +AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format. -**OmniRoute 如何解决:** +**How OmniRoute solves it:** --**系统提示注入**— 全局提示应用于所有请求 --**思考预算验证**- 每个请求的推理令牌分配控制(直通、自动、自定义、自适应) --**9 路由策略**— 确定如何分发请求的全局策略 --**通配符路由器**— `provider/*` 模式动态路由到任何提供商 --**组合启用/禁用切换**— 直接从仪表板切换组合 --**提供商切换**— 一键启用/禁用提供商的所有连接 --**阻止的提供商**— 从“/v1/models”列表中排除特定提供商
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models +- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI) +- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI +- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen) +- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3 +- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers +- **Moderations** — `/v1/moderations` — Content safety checks +- **Reranking** — `/v1/rerank` — Document relevance reranking +- **Responses API** — Full `/v1/responses` support for Codex -<详情> -🧰 17.“我需要 MCP 工具作为一流的产品能力” + -许多 AI 网关仅将 MCP 作为隐藏的实现细节公开。团队需要一个可见的、可管理的操作层。 +
+🧪 14. "I have no way to test and compare quality across models" -**OmniRoute 如何解决:** +Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist. -- MCP 显示在仪表板导航和端点协议选项卡中 -- 专用 MCP 管理页面,包含流程、工具、范围和审计 -- 内置“omniroute --mcp”快速启动和客户端入门
+**How OmniRoute solves it:** -<详情> -🧠 18.“我需要具有同步+流任务路径的 A2A 编排” +- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal +- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function) +- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison +- **Chat Tester** — Full round-trip with visual response rendering +- **Live Monitor** — Real-time stream of all requests flowing through the proxy -代理工作流程需要直接回复和具有生命周期控制的长时间运行的流式执行。 + -**OmniRoute 如何解决:** +
+📈 15. "I need to scale without losing performance" -- 具有“消息/发送”和“消息/流”的 A2A JSON-RPC 端点(“POST /a2a”) -- 具有终端状态传播的 SSE 流式传输 -- 用于“tasks/get”和“tasks/cancel”的任务生命周期 API
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected. -<详情> -🛰️ 19.“我需要真实的 MCP 进程运行状况,而不是猜测的状态” +**How OmniRoute solves it:** -运营团队需要知道 MCP 是否确实存在,而不仅仅是 API 是否可访问。 +- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency +- **Request Idempotency** — 5s deduplication window for identical requests +- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking +- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence +- **API Key Validation Cache** — 3-tier cache for production performance +- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime -**OmniRoute 如何解决:** + -- 带有 PID、时间戳、传输、工具计数和范围模式的运行时心跳文件 -- MCP状态API结合心跳+最近的活动 -- 用于流程/正常运行时间/心跳新鲜度的 UI 状态卡 +
+🤖 16. "I want to control model behavior globally" -<详情> -📋 20.“我需要可审核的 MCP 工具执行” +Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical. -当工具改变配置或触发操作操作时,团队需要取证可追溯性。 +**How OmniRoute solves it:** -**OmniRoute 如何解决:** +- **System Prompt Injection** — Global prompt applied to all requests +- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive) +- **9 Routing Strategies** — Global strategies that determine how requests are distributed +- **Wildcard Router** — `provider/*` patterns route dynamically to any provider +- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard +- **Provider Toggle** — Enable/disable all connections for a provider with one click +- **Blocked Providers** — Exclude specific providers from `/v1/models` listing -- SQLite 支持的 MCP 工具调用审核日志记录 -- 按工具、成功/失败、API 密钥和分页过滤 -- 仪表板审核表+自动化统计端点
+ -<详情> -🔐 21.“每次集成我都需要范围内的 MCP 权限” +
+🧰 17. "I need MCP tools as first-class product capabilities" -不同的客户端应该具有对工具类别的最低权限访问权限。 +Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer. -**OmniRoute 如何解决:** +**How OmniRoute solves it:** -- 10 个粒度 MCP 范围,用于受控工具访问 -- MCP 管理 UI 中的范围执行和可见性 -- 操作工具的安全默认姿势
+- MCP appears in the dashboard navigation and endpoint protocol tab +- Dedicated MCP management page with process, tools, scopes, and audit +- Built-in quick-start for `omniroute --mcp` and client onboarding -<详情> -⚙️ 22.“我需要无需重新部署的操作控制” + -团队需要在事件或成本事件期间快速更改运行时。 +
+🧠 18. "I need A2A orchestration with sync + stream task paths" -**OmniRoute 如何解决:** +Agent workflows need both direct replies and long-running streamed execution with lifecycle control. -- 直接从 MCP 仪表板切换组合激活 -- 应用预定义策略包中的弹性配置文件 -- 从同一操作面板重置断路器状态
+**How OmniRoute solves it:** -<详情> -🔄 23.“我需要实时 A2A 任务生命周期可见性和取消” +- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream` +- SSE streaming with terminal state propagation +- Task lifecycle APIs for `tasks/get` and `tasks/cancel` -如果没有生命周期可见性,任务事件就很难分类。 + -**OmniRoute 如何解决:** +
+🛰️ 19. "I need real MCP process health, not guessed status" -- 任务列表/按状态/技能过滤并分页 -- 深入了解任务元数据、事件和工件 -- 任务取消端点和带有确认的 UI 操作
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable. -<详情> -🌊 24.“我需要 A2A 负载的活动流指标” +**How OmniRoute solves it:** -流媒体工作流程需要对并发和实时连接的操作洞察。 +- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode +- MCP status API combining heartbeat + recent activity +- UI status cards for process/uptime/heartbeat freshness -**OmniRoute 如何解决:** + -- 活动流计数器集成到 A2A 状态中 -- 最后任务时间戳和每个状态计数 -- 用于实时操作监控的 A2A 仪表板卡 +
+📋 20. "I need auditable MCP tool execution" -<详情> -🪪 25.“我需要为客户发现标准代理” +When tools mutate config or trigger ops actions, teams need forensic traceability. -外部客户端和协调器需要机器可读的元数据来进行引导。 +**How OmniRoute solves it:** -**OmniRoute 如何解决:** +- SQLite-backed audit logging for MCP tool calls +- Filters by tool, success/failure, API key, and pagination +- Dashboard audit table + stats endpoints for automation -- 代理卡暴露于`/.wellknown/agent.json` -- 管理 UI 中显示的能力和技能 -- A2A 状态 API 包括用于自动化的发现元数据
+ -<详情> -🧭 26.“我需要产品用户体验中的协议可发现性” +
+🔐 21. "I need scoped MCP permissions per integration" -如果用户无法发现协议表面,采用和支持质量就会下降。 +Different clients should have least-privilege access to tool categories. -**OmniRoute 如何解决:** +**How OmniRoute solves it:** -- 合并**端点**页面,其中包含代理、MCP、A2A 和 API 端点选项卡 -- MCP 和 A2A 的在线服务状态切换(在线/离线) -- 从概述到专用管理选项卡的链接
+- 10 granular MCP scopes for controlled tool access +- Scope enforcement and visibility in MCP management UI +- Safe default posture for operational tooling -<详情> -🧪 27.“我需要与真实客户端进行端到端协议验证” + -模拟测试不足以在发布前验证协议兼容性。 +
+⚙️ 22. "I need operational controls without redeploying" -**OmniRoute 如何解决:** +Teams need quick runtime changes during incidents or cost events. -- E2E 套件,可启动应用程序并使用真正的 MCP SDK 客户端传输 -- A2A 客户端测试发现、发送、流式传输、获取和取消流程 -- 针对 MCP 审计和 A2A 任务 API 交叉检查断言
+**How OmniRoute solves it:** -<详情> -📡 28.“我需要跨所有接口的统一可观察性” +- Switch combo activation directly from MCP dashboard +- Apply resilience profiles from pre-defined policy packs +- Reset circuit breaker state from the same operations panel -按协议分割可观察性会产生盲点和更长的 MTTR。 + -**OmniRoute 如何解决:** +
+🔄 23. "I need live A2A task lifecycle visibility and cancellation" -- 一个产品中的统一仪表板/日志/分析 -- 跨 OpenAI、MCP 和 A2A 层的运行状况 + 审计 + 请求遥测 -- 用于状态和自动化的操作 API
+Without lifecycle visibility, task incidents become hard to triage. -<详情> -💼 29.“我需要一个用于代理+工具+代理编排的运行时” +**How OmniRoute solves it:** -运行许多单独的服务会增加运营成本和故障模式。 +- Task listing/filtering by state/skill with pagination +- Drill-down on task metadata, events, and artifacts +- Task cancellation endpoint and UI action with confirmation -**OmniRoute 如何解决:** + -- 兼容 OpenAI 的代理、MCP 服务器和 A2A 服务器位于一个堆栈中 -- 共享身份验证、弹性、数据存储和可观察性 -- 所有交互界面上一致的策略模型 +
+🌊 24. "I need active stream metrics for A2A load" -<详情> -🚀 30.“我需要在没有胶水代码蔓延的情况下交付代理工作流程” +Streaming workflows require operational insight into concurrency and live connections. -拼接多个临时服务和脚本时,团队会失去速度。 +**How OmniRoute solves it:** -**OmniRoute 如何解决:** +- Active stream counters integrated into A2A status +- Last task timestamp and per-state counts +- A2A dashboard cards for real-time ops monitoring -- 客户端和代理的统一端点策略 -- 内置协议管理 UI 和烟雾验证路径 -- 生产就绪的基础(安全性、日志记录、弹性、备份)
+ + +
+🪪 25. "I need standard agent discovery for clients" + +External clients and orchestrators need machine-readable metadata for onboarding. + +**How OmniRoute solves it:** + +- Agent Card exposed at `/.well-known/agent.json` +- Capabilities and skills shown in management UI +- A2A status API includes discovery metadata for automation + +
+ +
+🧭 26. "I need protocol discoverability in the product UX" + +If users cannot discover protocol surfaces, adoption and support quality drop. + +**How OmniRoute solves it:** + +- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints +- Inline service status toggles (Online/Offline) for MCP and A2A +- Links from overview to dedicated management tabs + +
+ +
+🧪 27. "I need end-to-end protocol validation with real clients" + +Mock tests are not enough to validate protocol compatibility before release. + +**How OmniRoute solves it:** + +- E2E suite that boots app and uses real MCP SDK client transport +- A2A client tests for discovery, send, stream, get, and cancel flows +- Cross-check assertions against MCP audit and A2A tasks APIs + +
+ +
+📡 28. "I need unified observability across all interfaces" + +Splitting observability by protocol creates blind spots and longer MTTR. + +**How OmniRoute solves it:** + +- Unified dashboards/logs/analytics in one product +- Health + audit + request telemetry across OpenAI, MCP, and A2A layers +- Operational APIs for status and automation + +
+ +
+💼 29. "I need one runtime for proxy + tools + agent orchestration" + +Running many separate services increases operational cost and failure modes. + +**How OmniRoute solves it:** + +- OpenAI-compatible proxy, MCP server, and A2A server in one stack +- Shared auth, resilience, data store, and observability +- Consistent policy model across all interaction surfaces + +
+ +
+🚀 30. "I need to ship agentic workflows without glue-code sprawl" + +Teams lose velocity when stitching multiple ad-hoc services and scripts. + +**How OmniRoute solves it:** + +- Unified endpoint strategy for clients and agents +- Built-in protocol management UIs and smoke validation paths +- Production-ready foundations (security, logging, resilience, backup) + +
### Example Playbooks (Integrated Use Cases) -**剧本 A:最大化付费订阅 + 廉价备份**```txt +**Playbook A: Maximize paid subscription + cheap backup** + +```txt Combo: "maximize-claude" 1. cc/claude-opus-4-6 2. glm/glm-4.7 @@ -608,21 +690,23 @@ Combo: "maximize-claude" Monthly cost: $20 + small backup spend Outcome: higher quality, near-zero interruption -```` +``` -**剧本 B:零成本编码堆栈**```txt +**Playbook B: Zero-cost coding stack** + +```txt Combo: "free-forever" - -1. gc/gemini-3-flash -2. if/kimi-k2-thinking -3. qw/qwen3-coder-plus + 1. gc/gemini-3-flash + 2. if/kimi-k2-thinking + 3. qw/qwen3-coder-plus Monthly cost: $0 Outcome: stable free coding workflow +``` -```` +**Playbook C: 24/7 always-on fallback chain** -**剧本 C:24/7 始终在线的后备链**```txt +```txt Combo: "always-on" 1. cc/claude-opus-4-6 2. cx/gpt-5.2-codex @@ -631,122 +715,134 @@ Combo: "always-on" 5. if/kimi-k2-thinking Outcome: deep fallback depth for deadline-critical workloads -```` +``` -**剧本 D:使用 MCP + A2A 进行特工操作**```txt +**Playbook D: Agent ops with MCP + A2A** -1. Start MCP transport (`omniroute --mcp`) for tool-driven operations -2. Run A2A tasks via `message/send` and `message/stream` -3. Observe via /dashboard/endpoint (MCP and A2A tabs) -4. Toggle services via inline status controls - -```` +```txt +1) Start MCP transport (`omniroute --mcp`) for tool-driven operations +2) Run A2A tasks via `message/send` and `message/stream` +3) Observe via /dashboard/endpoint (MCP and A2A tabs) +4) Toggle services via inline status controls +``` --- ## 🆓 Start Free — Zero Configuration Cost -> 只需几分钟即可设置 AI 编码,费用为**0 美元/月**。连接这些免费帐户并使用内置的**Free Stack**组合。 +> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo. -|步骤|行动|供应商已解锁 | +| Step | Action | Providers Unlocked | | ---- | -------------------------------------------------- | ------------------------------------------------------------------ | -| 1 |连接**Kiro**(AWS Builder ID OAuth)|克劳德十四行诗 4.5、俳句 4.5 —**无限**| -| 2 |连接**Qoder**(Google OAuth) | kimi-k2-thinking、qwen3-coder-plus、deepseek-r1... —**无限**| -| 3 |连接**Qwen**(设备代码)| qwen3-coder-plus、qwen3-coder-flash... —**无限制**| -| 4 |连接**Gemini CLI**(Google OAuth) | gemini-3-flash、gemini-2.5-pro —**180K/月 免费**| -| 5 | `/dashboard/combos` →**免费堆栈 ($0)**模板 |自动循环所有免费提供商 | +| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** | +| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** | +| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** | +| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** | +| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically | -**将任何 IDE/CLI 指向:**`http://localhost:20128/v1` · API 密钥:`any-string` · 完成。 +**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done. ->**可选的额外覆盖范围(也是免费的):**Groq API 密钥(30 RPM 免费)、NVIDIA NIM(40 RPM 免费,70 多个模型)、Cerebras(1M tok/天)、LongCat API 密钥(50M 令牌/天!)、Cloudflare Workers AI(10K 神经元/天,50 多个模型)。## 快速开始 +> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models). + +## 快速开始 ### 1) Install and run ```bash npm install -g omniroute omniroute -```` +``` -> **pnpm 用户:**安装后运行 `pnpmapprove-builds -g` 以启用 `better-sqlite3` 和 `@swc/core` 所需的本机构建脚本: +> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`: > > ```bash -> pnpm install -g 全方位路由 -> pnpmapprove-builds -g # 选择所有包 → 批准 -> 全向路线 +> pnpm install -g omniroute +> pnpm approve-builds -g # Select all packages → approve +> omniroute > ``` -仪表板在“http://localhost:20128”打开,API 基本 URL 为“http://localhost:20128/v1”。 +Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`. -| 命令 | 描述 | -| ----------------------- | ---------------------------------------------------- | -| `全向` | 启动服务器(`PORT=20128`,API 和仪表板位于同一端口) | -| `omniroute --端口 3000` | 将规范/API 端口设置为 3000 | -| `omniroute --mcp` | 启动 MCP 服务器(stdio 传输) | -| `omniroute --no-open` | 不要自动打开浏览器 | -| `omniroute --help` | 显示帮助 | +| Command | Description | +| ----------------------- | ----------------------------------------------------------- | +| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) | +| `omniroute --port 3000` | Set canonical/API port to 3000 | +| `omniroute --mcp` | Start MCP server (stdio transport) | +| `omniroute --no-open` | Don't auto-open browser | +| `omniroute --help` | Show help | -可选的分割端口模式:```bash +Optional split-port mode: + +```bash PORT=20128 DASHBOARD_PORT=20129 omniroute - -# API: http://localhost:20128/v1 - +# API: http://localhost:20128/v1 # Dashboard: http://localhost:20129 - -```` +``` ### Long-Running Streaming Timeouts -对于大多数部署,您只需要: +For most deployments, you only need: -|变量|默认 |目的| -| ------------------------ | -------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------- | -| `REQUEST_TIMEOUT_MS` | `600000` |上游获取的共享基线、隐藏的 Undici 超时、TLS 指纹请求和 API 桥接请求/代理超时 | -| `STREAM_IDLE_TIMEOUT_MS` |继承`REQUEST_TIMEOUT_MS` | OmniRoute 中止 SSE 流之前流块之间的最大间隙 | +| Variable | Default | Purpose | +| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts | +| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream | -保留向后兼容性:现有的“FETCH_TIMEOUT_MS”、“API_BRIDGE_PROXY_TIMEOUT_MS”和其他每层超时变量仍然有效并覆盖共享基线。 +Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline. -如果您需要更精细的控制,可以使用高级覆盖:|变量|默认 |目的| -| ---------------------------------------------------- | ------------------------------------------------------ | -------------------------------------------------------------------------------- | -| `FETCH_TIMEOUT_MS` |继承`REQUEST_TIMEOUT_MS` |主获取中止信号使用的总上游请求超时 | -| `FETCH_HEADERS_TIMEOUT_MS` |继承`FETCH_TIMEOUT_MS` | Undici 接收上游响应标头的时间限制 | -| `FETCH_BODY_TIMEOUT_MS` |继承`FETCH_TIMEOUT_MS` |上游主体块之间的 Undici 时间限制(“0”禁用它)| -| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP 连接超时 | -| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici 空闲保持活动套接字超时 | -| `TLS_CLIENT_TIMEOUT_MS` |继承`FETCH_TIMEOUT_MS` |通过“wreq-js”发出的 TLS 指纹请求超时 | -| `API_BRIDGE_PROXY_TIMEOUT_MS` |继承`REQUEST_TIMEOUT_MS`或`30000` |从 API 端口到仪表板端口的“/v1”代理转发超时 | -| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `最大(API_BRIDGE_PROXY_TIMEOUT_MS,300000)` | API 桥接服务器上的传入请求超时 | -| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | API 桥接服务器上的传入标头超时 | -| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | API 桥接服务器上的保持活动超时 | -| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | API 桥接服务器上的套接字不活动超时(“0”禁用它)| +Advanced overrides are available if you need finer control: -如果您在 Nginx、Caddy、Cloudflare 或其他反向代理后面运行 OmniRoute,请确保代理 -超时也高于 OmniRoute 流/获取超时。### 2) Connect providers and create your API key +| Variable | Default | Purpose | +| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- | +| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal | +| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers | +| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) | +| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout | +| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout | +| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` | +| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port | +| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server | +| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server | +| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server | +| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) | -1. 打开仪表板 → `Providers` 并连接至少一个提供商(OAuth 或 API 密钥)。 -2. 打开仪表板 → `Endpoints` 并创建 API 密钥。 -3.(可选)打开仪表板 → `Combos` 并设置后备链。### 3) Point your coding tool to OmniRoute +If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy +timeouts are also higher than your OmniRoute stream/fetch timeouts. + +### 2) Connect providers and create your API key + +1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key). +2. Open Dashboard → `Endpoints` and create an API key. +3. (Optional) Open Dashboard → `Combos` and set your fallback chain. + +### 3) Point your coding tool to OmniRoute ```txt Base URL: http://localhost:20128/v1 API Key: [copy from Endpoint page] Model: if/kimi-k2-thinking (or any provider/model prefix) -```` +``` -可与 Claude Code、Codex CLI、Gemini CLI、Cursor、Cline、OpenClaw、OpenCode 和 OpenAI 兼容的 SDK 配合使用。### 4) Enable and validate protocols (v2.0) +Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs. -**MCP(用于工具驱动的操作):**```bash +### 4) Enable and validate protocols (v2.0) + +**MCP (for tool-driven operations):** + +```bash omniroute --mcp +``` -```` - -然后通过“stdio”连接您的 MCP 客户端并测试工具,例如: +Then connect your MCP client over `stdio` and test tools like: - `omniroute_get_health` - `omniroute_list_combos` -**A2A(针对客服人员到客服人员的工作流程):**```bash +**A2A (for agent-to-agent workflows):** + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` ```bash curl -X POST http://localhost:20128/a2a \ @@ -760,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \ npm run test:protocols:e2e ``` -该套件根据正在运行的应用程序验证真实的 MCP 和 A2A 客户端流。### Alternative: run from source +This suite validates real MCP and A2A client flows against a running app. + +### Alternative: run from source ```bash cp .env.example .env @@ -768,14 +866,13 @@ npm install PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev ``` -<详情> +
+Void Linux (`xbps-src` template) -Void Linux(`xbps-src`模板) - -对于 Void Linux 用户,您可以使用“xbps-src”构建本机包。将此块保存为“srcpkgs/omniroute/template”:```bash +For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`: +```bash # Template file for 'omniroute' - pkgname=omniroute version=3.4.1 revision=1 @@ -787,7 +884,7 @@ license="MIT" homepage="https://github.com/diegosouzapw/OmniRoute" distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz" checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b -system_accounts="\_omniroute" +system_accounts="_omniroute" omniroute_homedir="/var/lib/omniroute" export NODE_ENV=production export npm_config_engine_strict=false @@ -795,71 +892,70 @@ export npm_config_loglevel=error export npm_config_fund=false export npm_config_audit=false -do_build() { # Determine target CPU arch for node-gyp -local \_gyp_arch -case "$XBPS_TARGET_MACHINE" in -aarch64*) \_gyp_arch=arm64 ;; -armv7*|armv6*) \_gyp_arch=arm ;; -i686*) \_gyp_arch=ia32 ;; -\*) \_gyp_arch=x64 ;; -esac +do_build() { + # Determine target CPU arch for node-gyp + local _gyp_arch + case "$XBPS_TARGET_MACHINE" in + aarch64*) _gyp_arch=arm64 ;; + armv7*|armv6*) _gyp_arch=arm ;; + i686*) _gyp_arch=ia32 ;; + *) _gyp_arch=x64 ;; + esac - # 1) Install all deps – skip scripts (no network in do_build, native modules - # compiled separately below; better-sqlite3 is serverExternalPackage so - # Next.js does not execute it during next build) - NODE_ENV=development npm ci --ignore-scripts + # 1) Install all deps – skip scripts (no network in do_build, native modules + # compiled separately below; better-sqlite3 is serverExternalPackage so + # Next.js does not execute it during next build) + NODE_ENV=development npm ci --ignore-scripts - # 2) Build the Next.js standalone bundle - npm run build + # 2) Build the Next.js standalone bundle + npm run build - # 3) Copy static assets into standalone - cp -r .next/static .next/standalone/.next/static - [ -d public ] && cp -r public .next/standalone/public || true + # 3) Copy static assets into standalone + cp -r .next/static .next/standalone/.next/static + [ -d public ] && cp -r public .next/standalone/public || true - # 4) Compile better-sqlite3 native binding for the target architecture. - # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used - # without npm altering them. - local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js - (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") + # 4) Compile better-sqlite3 native binding for the target architecture. + # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used + # without npm altering them. + local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js + (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch") - # 5) Place the compiled binding into the standalone bundle - local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release - mkdir -p "$_bs3_release" - cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" + # 5) Place the compiled binding into the standalone bundle + local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release + mkdir -p "$_bs3_release" + cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/" - # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true - # so sharp is not used at runtime; x64 .so files would break aarch64 strip - rm -rf .next/standalone/node_modules/@img - - # 7) Copy pino runtime deps omitted by Next.js static analysis: - # pino-abstract-transport – required by pino's worker thread - # split2 – dep of pino-abstract-transport - # process-warning – dep of pino itself - for _mod in pino-abstract-transport split2 process-warning; do - cp -r "node_modules/$_mod" .next/standalone/node_modules/ - done + # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true + # so sharp is not used at runtime; x64 .so files would break aarch64 strip + rm -rf .next/standalone/node_modules/@img + # 7) Copy pino runtime deps omitted by Next.js static analysis: + # pino-abstract-transport – required by pino's worker thread + # split2 – dep of pino-abstract-transport + # process-warning – dep of pino itself + for _mod in pino-abstract-transport split2 process-warning; do + cp -r "node_modules/$_mod" .next/standalone/node_modules/ + done } do_check() { -npm run test:unit + npm run test:unit } do_install() { -vmkdir usr/lib/omniroute/.next + vmkdir usr/lib/omniroute/.next - vcopy .next/standalone/. usr/lib/omniroute/.next/standalone + vcopy .next/standalone/. usr/lib/omniroute/.next/standalone - # Prevent removal of empty Next.js app router dirs by the post-install hook - for _d in \ - .next/standalone/.next/server/app/dashboard \ - .next/standalone/.next/server/app/dashboard/settings \ - .next/standalone/.next/server/app/dashboard/providers; do - touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" - done - - cat > "${WRKDIR}/omniroute" <<'EOF' + # Prevent removal of empty Next.js app router dirs by the post-install hook + for _d in \ + .next/standalone/.next/server/app/dashboard \ + .next/standalone/.next/server/app/dashboard/settings \ + .next/standalone/.next/server/app/dashboard/providers; do + touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep" + done + cat > "${WRKDIR}/omniroute" <<'EOF' #!/bin/sh export PORT="${PORT:-20128}" export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}" @@ -871,10 +967,9 @@ EOF } post_install() { -vlicense LICENSE + vlicense LICENSE } - -```` +```
@@ -882,9 +977,11 @@ vlicense LICENSE ## 🐳 Docker -OmniRoute 在 [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute) 上作为公共 Docker 映像提供。 +OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute). -**快速运行:**```bash +**Quick run:** + +```bash docker run -d \ --name omniroute \ --restart unless-stopped \ @@ -892,24 +989,23 @@ docker run -d \ -p 20128:20128 \ -v omniroute-data:/app/data \ diegosouzapw/omniroute:latest -```` +``` -**带有环境文件:**```bash +**With environment file:** +```bash # Copy and edit .env first - cp .env.example .env docker run -d \ - --name omniroute \ - --restart unless-stopped \ - --stop-timeout 40 \ - --env-file .env \ - -p 20128:20128 \ - -v omniroute-data:/app/data \ - diegosouzapw/omniroute:latest - -```` + --name omniroute \ + --restart unless-stopped \ + --stop-timeout 40 \ + --env-file .env \ + -p 20128:20128 \ + -v omniroute-data:/app/data \ + diegosouzapw/omniroute:latest +``` **Using Docker Compose:** @@ -919,60 +1015,70 @@ docker compose --profile base up -d # CLI profile (Claude Code, Codex, OpenClaw built-in) docker compose --profile cli up -d -```` +``` -对 Docker 部署的仪表板支持现在包括“仪表板 → 端点”上的一键**Cloudflare 快速隧道**。第一个启用仅在需要时下载“cloudflared”,启动到当前“/v1”端点的临时隧道,并在正常公共 URL 正下方显示生成的“https://\*.trycloudflare.com/v1” URL。 +Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL. -注意事项: +Notes: -- 快速隧道 URL 是临时的,每次重新启动后都会更改。 -- OmniRoute 或容器重新启动后,快速隧道不会自动恢复。需要时从仪表板重新启用它们。 -- 托管安装目前支持“x64”/“arm64”上的 Linux、macOS 和 Windows。 -- 托管快速隧道默认采用 HTTP/2 传输,以避免在受限容器环境中出现嘈杂的 QUIC UDP 缓冲区警告。如果您想要不同的传输,请设置“CLOUDFLARED_PROTOCOL=quic”或“auto”。 -- Docker 镜像捆绑系统 CA 根并将其传递给托管的“cloudflared”,这样可以避免在容器内引导隧道时出现 TLS 信任失败。 -- SQLite 以 WAL 模式运行。应允许“docker stop”完成,以便 OmniRoute 可以将最新更改检查点回“storage.sqlite”。 -- 捆绑的 Compose 文件已设置 40 秒的停止宽限期。如果直接运行映像,请保留“--stop-timeout 40”(或类似值),以便手动停止不会中断关机清理。 -- 如果您希望 OmniRoute 使用现有二进制文件而不是下载二进制文件,请设置“CLOUDFLARED_BIN=/absolute/path/to/cloudflared”。 +- Quick Tunnel URLs are temporary and change after every restart. +- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed. +- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`. +- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport. +- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container. +- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. +- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup. +- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one. -**将 Docker Compose 与 Caddy 结合使用(HTTPS 自动 TLS):** +**Using Docker Compose with Caddy (HTTPS Auto-TLS):** -可以使用 Caddy 的自动 SSL 配置安全地公开 OmniRoute。确保您域的 DNS A 记录指向您服务器的 IP。```yaml +OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP. + +```yaml services: -omniroute: -image: diegosouzapw/omniroute:latest -container_name: omniroute -restart: unless-stopped -volumes: - omniroute-data:/app/data -environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com + omniroute: + image: diegosouzapw/omniroute:latest + container_name: omniroute + restart: unless-stopped + volumes: + - omniroute-data:/app/data + environment: + - PORT=20128 + - NEXT_PUBLIC_BASE_URL=https://your-domain.com -caddy: -image: caddy:latest -container_name: caddy -restart: unless-stopped -ports: - "80:80" - "443:443" -command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 + caddy: + image: caddy:latest + container_name: caddy + restart: unless-stopped + ports: + - "80:80" + - "443:443" + command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128 volumes: -omniroute-data: + omniroute-data: +``` -```` +| Image | Tag | Size | Description | +| ------------------------ | -------- | ------ | --------------------- | +| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release | +| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version | -|图片|标签 |尺寸|描述 | -| ------------------------ | -------- | ------ | -------------------- | -| `diegosouzapw/omniroute` | `最新` | 〜250MB |最新稳定版本 | -| `diegosouzapw/omniroute` | `1.0.3` | 〜250MB |当前版本 |--- +--- ## 🖥️ Desktop App — Offline & Always-On -> 🆕**新!**OmniRoute 现已作为适用于 Windows、macOS 和 Linux 的**本机桌面应用程序**提供。 +> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux. -将 OmniRoute 作为独立的桌面应用程序运行 — 本地模型无需终端、浏览器、互联网。基于 Electron 的应用程序包括: +Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes: -- 🖥️**本机窗口**— 具有系统托盘集成的专用应用程序窗口 -- 🔄**自动启动**— 在系统登录时启动 OmniRoute -- 🔔**本机通知**— 获取配额耗尽或提供商问题的警报 -- ⚡**一键安装**— NSIS (Windows)、DMG (macOS)、AppImage (Linux) -- 🌐**离线模式**— 使用捆绑服务器完全离线工作### 快速开始 +- 🖥️ **Native Window** — Dedicated app window with system tray integration +- 🔄 **Auto-Start** — Launch OmniRoute on system login +- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues +- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux) +- 🌐 **Offline Mode** — Works fully offline with bundled server + +### 快速开始 ```bash # Development mode @@ -983,308 +1089,370 @@ npm run electron:build # Current platform npm run electron:build:win # Windows (.exe) npm run electron:build:mac # macOS (.dmg) — x64 & arm64 npm run electron:build:linux # Linux (.AppImage) -```` +``` ### System Tray -最小化后,OmniRoute 会出现在您的系统托盘中,并可进行快速操作: +When minimized, OmniRoute lives in your system tray with quick actions: -- 打开仪表板 -- 更改服务器端口 -- 退出应用程序 +- Open dashboard +- Change server port +- Quit application -📖 完整文档:[`electron/README.md`](electron/README.md)--- +📖 Full documentation: [`electron/README.md`](electron/README.md) + +--- ## 💰 Pricing at a Glance -|等级 |供应商|成本|配额重置 |最适合 | -| ------------------- | ------------------------ | | ---------------------------------- | ---------------- | --------------------------------- | -|**💳 订阅**|克劳德代码(专业版)| $20/月 | 5 小时+ 每周 |已经订阅 | -| | Codex(增强版/专业版)| $20-200/月 | 5 小时+ 每周 | OpenAI 用户 | -| |双子座 CLI |**免费**| 180K/月 + 1K/天 |每个人! | -| | GitHub 副驾驶 | $10-19/月 |每月 | GitHub 用户 | -|**🔑 API 密钥**| NVIDIA NIM |**免费**(永远开发)|约 40 转/分 | 70+ 开放模型 | -| |大脑 |**免费**(1M tok/天)| 60K TPM / 30 转/分钟 |世界最快 | -| |格罗克 |**免费**(30 RPM) | 14.4K RPD |超快 Llama/Gemma | -| | DeepSeek V3.2 |每 100 万美元 0.27 美元/1.10 美元 |无 |最佳价格/质量推理 | -| | xAI Grok-4 快速 |**每 100 万美元 0.20 美元/0.50 美元**🆕 |无 |最快+工具调用,超低| -| | xAI Grok-4(标准)|每 100 万美元 0.20 美元/1.50 美元 🆕 |无 | xAI 推理旗舰 | -| |米斯特拉尔|免费试用+付费|限价|欧洲人工智能 | -| |开放路由器|按使用付费 |无 |总计 100 多个型号 | -|**💰便宜**| GLM-5(来自 Z.AI)🆕 | 0.5 美元/100 万美元 |每日上午 10 点 | 128K输出,最新旗舰| -| | GLM-4.7 | 0.6 美元/100 万美元 |每日上午 10 点 |预算备份| -| | MiniMax M2.5 🆕 | 0.3 美元/100 万输入 | 5小时滚动|推理+代理任务| -| |迷你最大M2.1 | 0.2 美元/100 万美元 | 5小时滚动 |最便宜的选择| -| | Kimi K2.5(Moonshot API)🆕 |按使用付费 |无 |直接 Moonshot API 访问 | -| |基米K2 |每月 9 美元的公寓 | 10M 代币/月 |可预测的成本| -|**🆓 免费**|科德尔 |**$0**|无限| 5款无限 | -| |奎文 |**$0**|无限| 4款无限| -| |基罗 |**$0**|无限| Claude Sonnet/俳句(AWS Builder)| -| | LongCat Flash-Lite 🆕 |**$0**(50M tok/天 🔥) | 1 RPS |地球上最大的免费配额| -| |授粉 AI 🆕 |**$0**(无需钥匙)| 1 请求/15 秒 | GPT-5、克劳德、DeepSeek、Llama 4 | -| | Cloudflare Workers AI 🆕 |**$0**(10K 神经元/天)| ~150 次/天 | 50+车型,全球优势| -| | Scaleway 人工智能 🆕 |**$0**(总计 100 万代币)|限价|欧盟/GDPR、Qwen3 235B、Llama 70B |> 🆕**添加新型号(2026 年 3 月):**Grok-4 Fast 系列售价 0.20 美元/0.50 美元/月(基准测试速度为 1143 毫秒 — 比 Gemini 2.5 Flash 快 30%)、GLM-5 通过 Z.AI 提供 128K 输出、MiniMax M2.5 推理、DeepSeek V3.2 更新定价、Kimi K2.5 通过 Moonshot 直接 API。 +| Tier | Provider | Cost | Quota Reset | Best For | +| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- | +| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed | +| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users | +| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! | +| | GitHub Copilot | $10-19/mo | Monthly | GitHub users | +| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models | +| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest | +| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma | +| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning | +| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow | +| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI | +| | Mistral | Free trial + paid | Rate limited | European AI | +| | OpenRouter | Pay-per-use | None | 100+ models aggr. | +| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship | +| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup | +| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks | +| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option | +| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access | +| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost | +| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited | +| | Qwen | **$0** | Unlimited | 4 models unlimited | +| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) | +| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth | +| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 | +| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge | +| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B | -**💡 0 美元组合堆栈 — 完整的免费设置:**``` +> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API. +**💡 $0 Combo Stack — The Complete Free Setup:** + +``` # 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever +Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED +Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key +Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day +Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day +NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +``` -Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED -Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED -LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 -Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed -Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED -Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key -Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day -Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) -Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day -NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever -Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever. -```` - -**零成本。永远不会停止编码。**将其配置为一个 OmniRoute 组合,所有回退都会自动发生 - 无需手动切换。--- +--- --- ## 🆓 Free Models — What You Actually Get -> 以下所有型号**100% 免费,零信用卡要求**。当一个配额用完时,OmniRoute 会在它们之间自动路由 - 将它们全部组合起来,形成牢不可破的 0 美元组合。### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) +> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo. -|型号|前缀|限制|速率限制 | -| ------------------- | ------ | ------------- | -------------------- | -| `克劳德十四行诗-4.5` | `kr/` |**无限制**|没有报道每日上限 | -| `克劳德俳句-4.5` | `kr/` |**无限制**|没有报道每日上限 | -| `克劳德-opus-4.6` | `kr/` |**无限制**| Kiro 的最新作品 |### 🟢 QODER MODELS (Free PAT via qodercli) +### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID) -|型号|前缀|限制|速率限制 | -| ------------------ | ------ | ------------- | ---------------- | -| `kimi-k2-思考` | `如果/` |**无限制**|没有报告上限 | -| `qwen3-coder-plus` | `如果/` |**无限制**|没有报告上限 | -| `deepseek-r1` | `如果/` |**无限制**|没有报告上限 | -| `minimax-m2.1` | `如果/` |**无限制**|没有报告上限 | -| `kimi-k2` | `如果/` |**无限制**|没有报告上限 | +| Model | Prefix | Limit | Rate Limit | +| ------------------- | ------ | ------------- | --------------------- | +| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap | +| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro | -> 推荐连接方法:**个人访问令牌 + `qodercli`**。浏览器 OAuth 是 -> 实验性的,默认禁用,除非配置了`QODER_OAUTH_*`环境变量。### 🟡 QWEN MODELS (Device Code Auth) +### 🟢 QODER MODELS (Free PAT via qodercli) -|型号|前缀|限制|速率限制 | +| Model | Prefix | Limit | Rate Limit | +| ------------------ | ------ | ------------- | --------------- | +| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap | +| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap | +| `deepseek-r1` | `if/` | **Unlimited** | No reported cap | +| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap | +| `kimi-k2` | `if/` | **Unlimited** | No reported cap | + +> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is +> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured. + +### 🟡 QWEN MODELS (Device Code Auth) + +| Model | Prefix | Limit | Rate Limit | | ------------------- | ------ | ------------- | ------------------- | -| `qwen3-coder-plus` | `qw/` |**无限制**|没有报告上限 | -| `qwen3-coder-flash` | `qw/` |**无限制**|没有报告上限 | -| `qwen3-coder-next` | `qw/` |**无限制**|没有报告上限 | -| `视觉模型` | `qw/` |**无限制**|多式联运(图像)|### 🟣 GEMINI CLI (Google OAuth) +| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap | +| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap | +| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) | -|型号|前缀|限制|速率限制 | -| ------------------------ | ------ | ------------------------ | | ------------- | -| `gemini-3-flash-预览` | `gc/` |**180K tok/月**+ 1K/天 |每月重置 | -| `gemini-2.5-pro` | `gc/` | 180K/月(共享池)|高品质|### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) +### 🟣 GEMINI CLI (Google OAuth) -|等级 |每日限额 |速率限制 |笔记| +| Model | Prefix | Limit | Rate Limit | +| ------------------------ | ------ | --------------------------- | ------------- | +| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset | +| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality | + +### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com) + +| Tier | Daily Limit | Rate Limit | Notes | | ---------- | ------------ | ----------- | ------------------------------------------------------ | -|免费(开发)|无代币上限 |**~40 转/分**| 70+型号; 2025 年中期过渡到纯粹的费率限制 | +| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 | -热门免费模型:`moonshotai/kimi-k2.5` (Kimi K2.5)、`z-ai/glm4.7` (GLM 4.7)、`deepseek-ai/deepseek-v3.2` (DeepSeek V3.2)、`nvidia/llama-3.3-70b-instruct`、`deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) +Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1` -|等级 |每日限额 |速率限制 |笔记| +### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai) + +| Tier | Daily Limit | Rate Limit | Notes | | ---- | ----------------- | ---------------- | ------------------------------------------- | -|免费|**100 万个代币/天**| 60K TPM / 30 转/分钟 |世界上最快的LLM推理;每日重置 | +| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily | -免费提供:`llama-3.3-70b`、`llama-3.1-8b`、`deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com) +Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b` -|等级 |每日限额 |速率限制 |笔记| -| ---- | ------------- | ---------------- | ---------------------------------------------------- | -|免费|**14.4K RPD**|每个型号 30 RPM |没有信用卡; 429 限量,不收费 | +### 🔴 GROQ (Free API Key — console.groq.com) -免费提供:`llama-3.3-70b-versatile`、`gemma2-9b-it`、`mixtral-8x7b`、`whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 +| Tier | Daily Limit | Rate Limit | Notes | +| ---- | ------------- | ---------------- | ----------------------------------------- | +| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged | -|型号|前缀|每日免费额度|笔记| -| -------------------------------------- | ------ | ----------------- | ----------------------- | -| `LongCat-Flash-Lite` | `lc/` |**5000 万代币**💥 |有史以来最大的免费配额| -| `LongCat-Flash-Chat` | `lc/` | 50 万个代币 |多轮聊天 | -| 《长猫闪思维》 | `lc/` | 50 万个代币 |推理/CoT | -| `LongCat-Flash-Thinking-2601` | `lc/` | 50 万个代币 | 2026 年 1 月版本 | -| `LongCat-Flash-Omni-2603` | `lc/` | 50 万个代币 |多式联运 | +Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3` -> 公测期间 100% 免费。使用电子邮件或电话在 [longcat.chat](https://longcat.chat) 上注册。每天 00:00 UTC 重置。### 🟢 POLLINATIONS AI (No API Key Required) 🆕 +### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕 -|型号|前缀|速率限制 |背后的提供商 | +| Model | Prefix | Daily Free Quota | Notes | +| ----------------------------- | ------ | ----------------- | ----------------------- | +| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever | +| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat | +| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT | +| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version | +| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal | + +> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC. + +### 🟢 POLLINATIONS AI (No API Key Required) 🆕 + +| Model | Prefix | Rate Limit | Provider Behind | | ---------- | ------ | ---------- | ------------------ | -| `openai` | `pol/` | 1 请求/15 秒 | GPT-5 | -| '克劳德' | `pol/` | 1 请求/15 秒 |人类克劳德 | -| `双子座` | `pol/` | 1 请求/15 秒 |谷歌双子座 | -| '深探' | `pol/` | 1 请求/15 秒 |深思V3 | -| `美洲驼` | `pol/` | 1 请求/15 秒 | Meta Llama 4 侦察兵 | -|米斯特拉尔 | `pol/` | 1 请求/15 秒 |米斯特拉尔人工智能 | +| `openai` | `pol/` | 1 req/15s | GPT-5 | +| `claude` | `pol/` | 1 req/15s | Anthropic Claude | +| `gemini` | `pol/` | 1 req/15s | Google Gemini | +| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 | +| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout | +| `mistral` | `pol/` | 1 req/15s | Mistral AI | -> ✨**零摩擦:**无需注册,无需 API 密钥。添加具有空键字段的授粉提供程序,它会立即起作用。### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 +> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately. -|等级 |每日神经元 |等效用法|笔记| +### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕 + +| Tier | Daily Neurons | Equivalent Usage | Notes | | ---- | ------------- | --------------------------------------- | ----------------------- | -|免费|**10,000**|约 150 个法学硕士 / 500 秒音频 / 15K 嵌入 |全球优势,50+型号| +| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models | -流行的免费模型:`@cf/meta/llama-3.3-70b-instruct`、`@cf/google/gemma-3-12b-it`、`@cf/openai/whisper-large-v3-turbo`(免费音频!)、`@cf/qwen/qwen2.5-coder-15b-instruct` +Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct` -> 需要来自 [dash.cloudflare.com](https://dash.cloudflare.com) 的 API 令牌 + 帐户 ID。将帐户 ID 存储在提供商设置中。### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 +> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings. -|等级 |免费配额 |地点 |笔记| +### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕 + +| Tier | Free Quota | Location | Notes | | ---- | ------------- | ------------ | ----------------------------------- | -|免费|**1M 代币**| 🇫🇷 欧盟巴黎 |限额内无需信用卡 | +| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits | -免费提供:“qwen3-235b-a22b-instruct-2507”(Qwen3 235B!)、“llama-3.1-70b-instruct”、“mistral-small-3.2-24b-instruct-2506”、“deepseek-v3-0324” +Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324` -> 符合欧盟/GDPR 标准。在 [console.scaleway.com](https://console.scaleway.com) 获取 API 密钥。 +> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com). ->**💡 终极免费堆栈(11 个提供商,永远 0 美元):** +> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):** > > ``` -> Kiro (kr/) → 克劳德十四行诗/俳句无限 -> Qoder (if/) → kimi-k2-thinking、qwen3-coder-plus、deepseek-r1 无限 -> LongCat Lite (lc/) → LongCat-Flash-Lite — 5000 万代币/天 🔥 -> 授粉 (pol/) → GPT-5、Claude、DeepSeek、Llama 4 — 无需密钥 -> Qwen (qw/) → qwen3-coder 模型无限 -> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 请求/天免费 -> Cloudflare AI (cf/) → 50 多个模型 — 10K 神经元/天 -> Scaleway (scw/) → Qwen3 235B、Llama 70B — 100 万个免费代币(欧盟) -> Groq (groq/) → Llama/Gemma — 14.4K 请求/天超快 -> NVIDIA NIM (nvidia/) → 70 多个开放型号 — 永远 40 RPM -> Cerebras (cerebras/) → Llama/Qwen 世界最快 — 1M tok/天 -> ```## 🎙️ Free Transcription Combo +> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED +> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED +> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥 +> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed +> Qwen (qw/) → qwen3-coder models UNLIMITED +> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free +> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day +> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU) +> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast +> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever +> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day +> ``` -> 以**0 美元**转录任何音频/视频 — Deepgram 领先,免费 200 美元,AssemblyAI 50 美元后备,Groq Whisper 作为无限紧急备份。 +## 🎙️ Free Transcription Combo -|供应商|免费积分|最佳模特|速率限制 | -| ----------------- | ---------------------- | -------------------------------------------------------- | ---------------------------- | -| 🟢**Deepgram**|**200 美元免费**(注册)| `nova-3` — 最高准确度,30 多种语言 |免费积分没有 RPM 限制 | -| 🔵**AssemblyAI**|**50 美元免费**(注册)| `universal-3-pro` — 章节、情绪、PII |免费积分没有 RPM 限制 | -| 🔴**Groq**|**永远免费**| `whisper-large-v3` — OpenAI Whisper | 30 RPM(速率有限)| +> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup. -**`/dashboard/combos`中建议的组合:**``` +| Provider | Free Credits | Best Model | Rate Limit | +| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- | +| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits | +| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits | +| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) | + +**Suggested combo in `/dashboard/combos`:** + +``` Name: free-transcription Strategy: Priority Nodes: [1] deepgram/nova-3 → uses $200 free first [2] assemblyai/universal-3-pro → fallback when Deepgram credits run out [3] groq/whisper-large-v3 → free forever, emergency fallback -```` +``` -然后在 `/dashboard/media` →**转录**选项卡中:上传任何音频或视频文件 → 选择您的组合端点 → 以支持的格式获取转录。## 💡 Key Features +Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats. -OmniRoute v2.0 被构建为一个操作平台,而不仅仅是一个中继代理。### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) +## 💡 Key Features -| 特色 | 它有什么作用 | -| ---------------------------- | ------------------------------------------------------------------------------- | ------------------------------------------------------------ | -| ⚡**Grok-4 快速家族** | xAI 模型价格为 $0.20/$0.50/M — 基准测试为 1143ms(比 Gemini 2.5 Flash 快 30%) | -| 🧠**GLM-5 通过 Z.AI** | 128K 输出上下文,0.5 美元/100 万美元 — GLM 系列的最新旗舰产品 | -| 🔮**MiniMax M2.5** | 推理 + 代理任务,价格为 0.30 美元/100 万美元 — M2.1 的重大升级 | -| 🎯**每个模型的工具调用标志** | 注册表中的每个模型 `toolCalling: true/false` — AutoCombo 会跳过不支持工具的模型 | -| 🌍**多语言意图检测** | AutoCombo 评分中的 PT/ZH/ES/AR 关键字 — 更好地选择非英语内容的模型 | -| 📊**基准驱动的回退** | 来自实时请求的真实 p95 延迟提供组合评分 — AutoCombo 从实际数据中学习 | -| 🔁**请求重复数据删除** | 基于内容哈希的重复数据删除窗口 — 多代理安全,防止重复收费 | -| 🔌**可插拔路由器策略** | 可扩展的“RouterStrategy”接口 - 添加自定义路由逻辑作为插件 | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP | +OmniRoute v3.5 is built as an operational platform, not just a relay proxy. -| 特色 | 它有什么作用 | -| --------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- | -| 🎮**模型游乐场** | 直接测试任何模型的仪表板页面 - 提供者/模型/端点选择器、Monaco 编辑器、流式传输、中止、计时 | -| 🔏**CLI 指纹匹配** | 每个提供商的标头/正文排序以匹配本机 CLI 签名 - 在“设置”>“安全”中切换每个提供商。**您的代理 IP 已保留** | -| 🤝**ACP 支持(代理客户端协议)** | CLI 代理发现(Codex、Claude、Goose、Gemini CLI、OpenClaw + 9 个以上)、进程生成器、`/api/acp/agents` 端点 | -| 🤖**ACP 代理仪表板** | 调试 › 代理页面 — 14 个代理网格,包含任何 CLI 工具的安装状态、版本、自定义代理表单。**OpenCode**用户可以获得一个“下载 opencode.json”按钮,该按钮会自动生成包含所有可用模型的即用型配置。 | -| 🔧**自定义模型 `apiFormat` 路由** | 带有 `apiFormat: "responses"` 的自定义模型现在可以正确路由到 Responses API 转换器 | -| 🏢**Codex 工作区隔离** | 每封电子邮件有多个 Codex 工作区 — OAuth 通过工作区 ID 正确分隔连接 | -| 🔄**Electron 自动更新** | 桌面应用程序检查更新+重新启动时自动安装 | ### 🤖 Agent & Protocol Operations (v2.0) | +### 🆕 New — v3.5.5 Highlights (Apr 2026) -| 特色 | 它有什么作用 | -| ---------------------------------- | ----------------------------------------------------------------------------------------------------------------- | ----------------------------- | -| 🔧**MCP 服务器(25 个工具)** | 通过 3 种传输的 IDE/代理工具:stdio、SSE (`/api/mcp/sse`)、可流式 HTTP (`/api/mcp/stream`)。 18核+3内存+4技能工具 | -| 🤝**A2A 服务器(JSON-RPC + SSE)** | 通过同步和流式传输代理到代理的任务执行 | -| 🧭**综合端点页面** | 带有端点代理、MCP、A2A 和 API 端点选项卡的选项卡式管理页面 | -| 🎚️**服务启用/禁用切换** | MCP 和 A2A 的 ON/OFF 开关,具有设置持久性(默认值:OFF) | -| 🛰️**MCP 运行时心跳** | 真实进程状态(pid、正常运行时间、心跳寿命、传输、范围模式) | -| 📋**MCP 审计追踪** | 可过滤的审核日志,包含成功/失败和关键归因 | -| 🔐**MCP 范围执行** | 受控工具访问的 10 个精细范围权限 | -| 📡**A2A 任务生命周期管理** | 列出/过滤任务、检查事件/工件、取消正在运行的任务 | -| 📋**代理卡发现** | 用于客户端自动发现的`/.well-known/agent.json` | -| 🧪**协议 E2E 测试工具** | 真正的 MCP SDK + A2A 客户端在 `test:protocols:e2e` 中流动 | -| ⚙️**操作控制** | 开关组合、应用弹性配置文件、从一个控制表面重置断路器 | ### 🧠 Routing & Intelligence | +| Feature | What It Does | +| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- | +| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation | +| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs | +| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions | +| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection | +| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts | -| 特色 | 它有什么作用 | -| ---------------------- | ---------------------------------------------------- | ----------------------- | -| 🎯**智能 4 层回退** | 自动路由:订阅 → API 密钥 → 便宜 → 免费 | -| 📊**实时配额跟踪** | 每个提供商的实时代币计数 + 重置倒计时 | -| 🔄**格式翻译** | OpenAI ↔ Claude ↔ Gemini ↔ 具有模式安全转换的响应 | -| 👥**多帐户支持** | 每个提供商有多个帐户,可进行智能选择 | -| 🔄**自动令牌刷新** | OAuth 令牌通过重试自动刷新 | -| 🎨**自定义组合** | 9种平衡策略+后备链控制 | -| 🌐**通配符路由器** | `provider/*` 动态路由 | -| 🧠**思考预算控制** | 直通、自动、自定义和自适应推理限制 | -| 🔀**模型别名** | 内置+自定义模型别名和迁移安全 | -| ⚡**背景退化** | 将低优先级后台任务路由到更便宜的模型 | -| 🧪**任务感知智能路由** | 按内容类型自动选择模型(编码/视觉/分析/摘要) | -| 🔄**A2A 代理工作流程** | 用于状态多步骤代理执行的确定性 FSM 协调器 | -| 🔀**自适应路由** | 基于代币数量和提示复杂性的动态策略覆盖 | -| 🎲**提供商多元化** | 香农熵评分平衡自动组合流量分布 | -| 💬**系统提示注入** | 一致应用全球行为控制 | -| 📄**响应 API 兼容性** | 对 Codex 和高级代理工作流程的全面“/v1/responses”支持 | ### 🎵 Multi-Modal APIs | +### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026) -| 特色 | 它有什么作用 | -| ---------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | -| 🖼️**图像生成** | 具有云和本地后端的“/v1/images/ Generations” | -| 📐**嵌入** | 用于搜索和 RAG 管道的“/v1/embeddings” | -| 🎤**音频转录** | `/v1/audio/transcriptions` — 7 个提供商(Deepgram Nova 3、AssemblyAI、Groq Whisper、HuggingFace、ElevenLabs、OpenAI、Azure)、自动语言检测、MP4/MP3/WAV 支持 | -| 🔊**文本转语音** | `/v1/audio/speech` — 10 个提供商(ElevenLabs、OpenAI、Deepgram、Cartesia、PlayHT、HuggingFace、Nvidia NIM、I​​nworld、Coqui、Tortoise),并提供正确的错误消息 | -| 🎬**视频生成** | `/v1/videos/ Generations`(ComfyUI + SD WebUI 工作流程) | -| 🎵**音乐一代** | `/v1/music/generations`(ComfyUI 工作流程) | -| 🛡️**审核** | `/v1/moderations` 安全检查 | -| 🔀**重新排名** | `/v1/rerank` 用于相关性评分 | -| 🔍**网络搜索**🆕 | `/v1/search` — 5 个提供商(Serper、Brave、Perplexity、Exa、Tavily)、每月 6,500 多个免费、自动故障转移、缓存 | ### 🛡️ Resilience, Security & Governance | +| Feature | What It Does | +| ------------------------------------ | ------------------------------------------------------------------------------------------- | +| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) | +| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family | +| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 | +| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models | +| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content | +| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data | +| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges | +| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins | -| 特色 | 它有什么作用 | -| ----------------------------- | ---------------------------------------------------------------------- | -| 🔌**断路器** | 每个模型的行程/恢复与阈值控制 | -| 🎯**端点感知模型** | 自定义模型声明支持的端点+ API 格式 | -| 🛡️**抗雷群** | 重试/速率事件的互斥锁 + 信号量保护 | -| 🧠**语义 + 签名缓存** | 通过两个缓存层降低成本/延迟 | -| ⚡**请求幂等性** | 重复防护窗 | -| 🔒**TLS 指纹欺骗** | 类似浏览器的 TLS 指纹 —**减少机器人检测和帐户标记** | -| 🔏**CLI 指纹匹配** | 匹配本机 CLI 请求签名 —**降低禁令风险,同时保留代理 IP** | -| 🌐**IP 过滤** | 公开部署的允许列表/阻止列表控制 | -| 📊**可编辑的速率限制** | 可配置的全局/提供商级别的持久限制 | -| 📉**优雅的降级** | 保护核心网关运营的多层能力回退 | -| 📜**配置审计跟踪** | 基于差异的变更跟踪通过简单的回滚防止操作漂移 | -| ⏳**提供商健康同步** | 主动令牌过期监控在授权失败之前触发警报 | -| 🚪**自动禁用被禁止的帐户** | 运行断路器自动密封永久冻结的代币账户 | -| 🔑**API 密钥管理 + 范围界定** | 安全密钥发行/轮换和模型/提供商控制 | -| 👁️**范围 API 密钥公开**🆕 | 通过“ALLOW_API_KEY_REVEAL”选择恢复 API 密钥 | -| 🛡️**受保护的`/models`** | 模型目录的可选身份验证门控和提供者隐藏### 📊 Observability & Analytics | +### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP -| 特色 | 它有什么作用 | -| --------------------- | ---------------------------------------------- | ---------------------------- | -| 📝**请求 + 代理日志** | 完整的请求/响应和代理日志记录 | -| 📉**流式详细日志**🆕 | 将 SSE 负载流干净地重构到 UI 中 | -| 📋**统一日志仪表板** | 一页中的请求、代理、​​审核和控制台视图 | -| 🔍**请求遥测** | p50/p95/p99 延迟和请求跟踪 | -| 🏥**健康仪表板** | 正常运行时间、断​​路器状态、锁定、缓存统计信息 | -| 💰**成本跟踪** | 预算控制和每个型号的定价可见性 | -| 📈**分析可视化** | 模型/提供商使用情况洞察和趋势视图 | -| 🧪**评估框架** | 具有可配置匹配策略的黄金集测试 | -| 📡**实时诊断**🆕 | 用于精确组合实时测试的语义缓存旁路 | ### ☁️ Deployment & Platform | +| Feature | What It Does | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing | +| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** | +| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint | +| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. | +| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator | +| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID | +| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart | -| 特色 | 它有什么作用 | -| ----------------------- | ------------------------------------------------ | ---------------------- | -| 🌐**随处部署** | 本地主机、VPS、Docker、云环境 | -| 🚇**Cloudflare 隧道**🆕 | 从仪表板一键快速隧道集成 | -| 🔑**API 密钥模型过滤** | 通过分配的承载上下文角色过滤本机 /v1/models 响应 | -| ⚡**智能缓存绕过** | 可配置的 TTL 启发式和强制重新获取控制 | -| 🔄**备份/恢复** | 出口/进口和灾难恢复流程 | -| 🧙**入门向导** | 首次运行引导设置 | -| 🔧**CLI 工具仪表板** | 一键设置流行的编码工具 | -| 🎮**模型游乐场** | 从仪表板测试任何提供者/模型/端点 | -| 🔏**CLI 指纹切换** | 设置 > 安全 | 中每个提供商的指纹匹配 | -| 🌐**i18n(30 种语言)** | 完整的仪表板 + 文档语言支持和 RTL 覆盖 | -| 🧹**清除所有型号** | 供应商详情中一键清空型号列表 | -| 👁️**侧边栏控件**🆕 | 从外观设置中隐藏组件和集成 | -| 📋**问题模板** | 针对错误和功能的标准化 GitHub 模板 | -| 📂**自定义数据目录** | 存储位置的`DATA_DIR`覆盖### Feature Deep Dive | +### 🤖 Agent & Protocol Operations (v2.0) + +| Feature | What It Does | +| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools | +| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows | +| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs | +| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) | +| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) | +| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution | +| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access | +| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks | +| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery | +| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` | +| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface | + +### 🧠 Routing & Intelligence + +| Feature | What It Does | +| ---------------------------------- | ------------------------------------------------------------------------ | +| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free | +| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider | +| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions | +| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection | +| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry | +| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control | +| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session | +| 🌐 **Wildcard Router** | `provider/*` dynamic routing | +| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits | +| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety | +| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models | +| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) | +| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions | +| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity | +| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution | +| 💬 **System Prompt Injection** | Global behavior controls applied consistently | +| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows | + +### 🎵 Multi-Modal APIs + +| Feature | What It Does | +| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends | +| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines | +| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support | +| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages | +| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) | +| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) | +| 🛡️ **Moderations** | `/v1/moderations` safety checks | +| 🔀 **Reranking** | `/v1/rerank` for relevance scoring | +| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache | + +### 🛡️ Resilience, Security & Governance + +| Feature | What It Does | +| ----------------------------------- | -------------------------------------------------------------------------------------- | +| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls | +| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format | +| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events | +| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers | +| ⚡ **Request Idempotency** | Duplicate protection window | +| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** | +| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** | +| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments | +| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence | +| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations | +| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks | +| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures | +| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically | +| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls | +| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` | +| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog | + +### 📊 Observability & Analytics + +| Feature | What It Does | +| -------------------------------- | ----------------------------------------------------- | +| 📝 **Request + Proxy Logging** | Full request/response and proxy logging | +| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI | +| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page | +| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing | +| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats | +| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility | +| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views | +| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies | +| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing | + +### ☁️ Deployment & Platform + +| Feature | What It Does | +| ------------------------------ | --------------------------------------------------------------------- | +| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments | +| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard | +| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles | +| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls | +| 🔄 **Backup/Restore** | Export/import and disaster recovery flows | +| 🧙 **Onboarding Wizard** | First-run guided setup | +| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools | +| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard | +| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security | +| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage | +| 🧹 **Clear All Models** | One-click model list clearing in provider details | +| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings | +| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features | +| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location | + +### Feature Deep Dive #### Smart fallback with practical cost control @@ -1296,105 +1464,132 @@ Combo: "my-coding-stack" 4. if/kimi-k2-thinking ``` -当配额、费率或运行状况失败时,OmniRoute 会自动移至下一个候选,无需手动切换。#### Protocol management that is visible and operable +When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching. -- MCP + A2A 可在 UI 和文档中发现(未隐藏) -- 协议状态 API 公开实时操作数据(`/api/mcp/*`、`/api/a2a/*`) -- 仪表板包括第二天操作的操作(组合切换、断路器重置、任务取消)#### Translator + validation workflow +#### Protocol management that is visible and operable -翻译器区域包括: +- MCP + A2A are discoverable in UI and docs (not hidden) +- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`) +- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation) --**Playground**:请求转换检查 -**聊天测试器**:完整的请求/响应往返 -**测试台**:一次运行多个案例 -**实时监控**:实时交通视图 +#### Translator + validation workflow -另外,通过“npm run test:protocols:e2e”与真实客户端进行协议验证。 +The Translator area includes: -> 📖**[MCP 服务器自述文件](open-sse/mcp-server/README.md)**— 工具参考、IDE 配置和客户端示例 +- **Playground**: request transformation checks +- **Chat Tester**: full request/response round-trip +- **Test Bench**: multiple cases in one run +- **Live Monitor**: real-time traffic view + +Plus protocol validation with real clients via `npm run test:protocols:e2e`. + +> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples > -> 📖**[A2A 服务器自述文件](src/lib/a2a/README.md)**— 技能、JSON-RPC 方法、流式处理和任务生命周期## 🧪 Evaluations (Evals) +> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle -OmniRoute 包含一个内置评估框架,用于根据黄金集测试 LLM 响应质量。通过仪表板中的**分析→评估**访问它。### Built-in Golden Set +## 🧪 Evaluations (Evals) -预加载的“OmniRoute Golden Set”包含以下测试用例: +OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard. -- 问候、数学、地理、代码生成 -- JSON格式合规、翻译、Markdown生成 -- 安全拒绝(有害内容)、计数、布尔逻辑### Evaluation Strategies +### Built-in Golden Set -| 战略 | 描述 | 示例 | -| ------------ | ------------------------------------ | -------------------------- | --- | -| `精确` | 输出必须完全匹配 | `"4"` | -| `包含` | 输出必须包含子字符串(不区分大小写) | `“巴黎”` | -| `正则表达式` | 输出必须匹配正则表达式模式 | `"1.*2.*3"` | -| `定制` | 自定义 JS 函数返回 true/false | `(输出) => 输出.长度 > 10` | --- | +The pre-loaded "OmniRoute Golden Set" contains test cases for: + +- Greetings, math, geography, code generation +- JSON format compliance, translation, markdown generation +- Safety refusal (harmful content), counting, boolean logic + +### Evaluation Strategies + +| Strategy | Description | Example | +| ---------- | ------------------------------------------------ | -------------------------------- | +| `exact` | Output must match exactly | `"4"` | +| `contains` | Output must contain substring (case-insensitive) | `"Paris"` | +| `regex` | Output must match regex pattern | `"1.*2.*3"` | +| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` | + +--- ## 📖 Setup Guide ### Protocol Setup (MCP + A2A) -<详情> +
+🧩 MCP Setup (Model Context Protocol) -🧩 MCP 设置(模型上下文协议) +Start MCP transport in stdio mode: -以 stdio 模式启动 MCP 传输:```bash +```bash omniroute --mcp +``` -```` +Recommended validation flow: -推荐的验证流程: +1. Connect your MCP client over stdio. +2. Run `omniroute_get_health`. +3. Run `omniroute_list_combos`. +4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit. -1. 通过 stdio 连接 MCP 客户端。 -2. 运行 `omniroute_get_health`。 -3. 运行 `omniroute_list_combos`。 -4. 打开 `/dashboard/mcp` 以确认心跳、活动和审核。 - -有用的自动化 API: +Useful APIs for automation: - `GET /api/mcp/status` - `GET /api/mcp/tools` - `GET /api/mcp/audit` -- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats` -<详情> -🤝 A2A 设置(Agent2Agent) + -发现代理:```bash +
+🤝 A2A Setup (Agent2Agent) + +Discover the agent: + +```bash curl http://localhost:20128/.well-known/agent.json -```` +``` -发送任务:```bash +Send a task: + +```bash curl -X POST http://localhost:20128/a2a \ - -H 'content-type: application/json' \ - -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' + -H 'content-type: application/json' \ + -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}' +``` -```` - -管理生命周期: +Manage lifecycle: - `GET /api/a2a/status` - `GET /api/a2a/tasks` - `GET /api/a2a/tasks/:id` - `POST /api/a2a/tasks/:id/cancel` -操作界面: +Operational UI: -- `/dashboard/a2a` 用于任务/状态/流可观察性和烟雾操作
+- `/dashboard/a2a` for task/state/stream observability and smoke actions -<详情> -🧪 端到端协议验证 + -与真实客户端验证这两个协议:```bash +
+🧪 End-to-end protocol validation + +Validate both protocols with real clients: + +```bash npm run test:protocols:e2e -```` +``` -这验证了: +This verifies: -- MCP SDK客户端连接/列表/调用 -- A2A发现/发送/流/获取/取消 -- 在 MCP 审计和 A2A 任务管理 API 中交叉检查数据
+- MCP SDK client connect/list/call +- A2A discovery/send/stream/get/cancel +- Cross-check data in MCP audit and A2A task management APIs -<详情> + -💳 订阅提供商### Claude Code (Pro/Max) +
+💳 Subscription Providers + +### Claude Code (Pro/Max) ```bash Dashboard → Providers → Connect Claude Code @@ -1407,7 +1602,9 @@ Models: cc/claude-haiku-4-5-20251001 ``` -**专业提示:**使用 Opus 来完成复杂的任务,使用 Sonnet 来提高速度。 OmniRoute 跟踪每个模型的配额!### OpenAI Codex (Plus/Pro) +**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model! + +### OpenAI Codex (Plus/Pro) ```bash Dashboard → Providers → Connect Codex @@ -1421,20 +1618,22 @@ Models: #### Codex Account Limit Management (5h + Weekly) -现在,每个 Codex 帐户在“仪表板 -> 提供商”中都有策略切换: +Each Codex account now has policy toggles in `Dashboard -> Providers`: -- `5h`(开/关):强制执行 5 小时窗口阈值政策。 -- “每周”(开/关):强制执行每周窗口阈值政策。 -- 阈值行为:当启用的窗口达到 >=90% 使用率时,将跳过该帐户。 -- 轮换行为:OmniRoute 自动路由到下一个符合条件的 Codex 帐户。 -- 重置行为:当提供商“resetAt”时间过去后,该帐户将自动再次符合资格。 +- `5h` (ON/OFF): enforce the 5-hour window threshold policy. +- `Weekly` (ON/OFF): enforce the weekly window threshold policy. +- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped. +- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically. +- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically. -应用场景: +Scenarios: -- `5h ON` + `Weekly ON`:当任一窗口达到阈值时,帐户将被跳过。 -- `5h OFF` + `Weekly ON`:只有每周使用才能阻止帐户。 -- `5h ON` + `Weekly OFF`:仅使用 5 小时即可冻结帐户。 -- `resetAt` 通过:帐户自动重新进入轮换(无需手动重新启用)。### Gemini CLI (FREE 180K/month!) +- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold. +- `5h OFF` + `Weekly ON`: only weekly usage can block the account. +- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account. +- `resetAt` passed: account re-enters rotation automatically (no manual re-enable). + +### Gemini CLI (FREE 180K/month!) ```bash Dashboard → Providers → Connect Gemini CLI @@ -1446,7 +1645,9 @@ Models: gc/gemini-2.5-pro ``` -**最超值:**巨大的免费套餐!在付费等级之前使用此功能。### GitHub Copilot +**Best Value:** Huge free tier! Use this before paid tiers. + +### GitHub Copilot ```bash Dashboard → Providers → Connect GitHub @@ -1461,71 +1662,91 @@ Models:
-<详情> +
+🔑 API Key Providers -🔑 API 密钥提供程序### NVIDIA NIM (FREE developer access — 70+ models) +### NVIDIA NIM (FREE developer access — 70+ models) -1. 注册:[build.nvidia.com](https://build.nvidia.com) -2. 获取免费 API 密钥(包含 1000 个推理积分) -3. 仪表板 → 添加提供商 → NVIDIA NIM: - - API 密钥:`nvapi-your-key` +1. Sign up: [build.nvidia.com](https://build.nvidia.com) +2. Get free API key (1000 inference credits included) +3. Dashboard → Add Provider → NVIDIA NIM: + - API Key: `nvapi-your-key` -**型号:**`nvidia/llama-3.3-70b-instruct`、`nvidia/mistral-7b-instruct` 以及 50 多个型号 +**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more -**专业提示:**OpenAI 兼容 API — 与 OmniRoute 的格式翻译无缝协作!### DeepSeek +**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation! -1. 注册:[platform.deepseek.com](https://platform.deepseek.com) -2. 获取API密钥 -3. 仪表板 → 添加提供商 → DeepSeek +### DeepSeek -**模型:**`deepseek/deepseek-chat`、`deepseek/deepseek-coder`### Groq (Free Tier Available!) +1. Sign up: [platform.deepseek.com](https://platform.deepseek.com) +2. Get API key +3. Dashboard → Add Provider → DeepSeek -1. 注册:[console.groq.com](https://console.groq.com) -2. 获取API密钥(包括免费套餐) -3. 仪表板 → 添加提供商 → Groq +**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder` -**型号:**`groq/llama-3.3-70b`、`groq/mixtral-8x7b` +### Groq (Free Tier Available!) -**专业提示:**超快速推理 - 最适合实时编码!### OpenRouter (100+ Models) +1. Sign up: [console.groq.com](https://console.groq.com) +2. Get API key (free tier included) +3. Dashboard → Add Provider → Groq -1. 注册:[openrouter.ai](https://openrouter.ai) -2. 获取API密钥 -3. 仪表板 → 添加提供商 → OpenRouter +**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b` -**模型:**通过单个 API 密钥访问来自所有主要提供商的 100 多个模型。 +**Pro Tip:** Ultra-fast inference — best for real-time coding! -**仪表板行为:**OpenRouter 模型通过**可用模型**进行管理。手动添加、导入和自动同步都会更新同一列表。
+### OpenRouter (100+ Models) -<详情> +1. Sign up: [openrouter.ai](https://openrouter.ai) +2. Get API key +3. Dashboard → Add Provider → OpenRouter -💰廉价提供商(备份)### GLM-4.7 (Daily reset, $0.6/1M) +**Models:** Access 100+ models from all major providers through a single API key. -1、注册:【智普AI】(https://open.bigmodel.cn/) 2. 从 Coding Plan 获取 API 密钥 3. 仪表板 → 添加 API 密钥: +**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list. -- 提供者:`glm` -- API 密钥:`你的密钥` + -**使用:**`glm/glm-4.7` +
+💰 Cheap Providers (Backup) -**专业提示:**Coding Plan 以 1/7 的成本提供 3× 配额!每天上午 10:00 重置。### MiniMax M2.1 (5h reset, $0.20/1M) +### GLM-4.7 (Daily reset, $0.6/1M) -1. 注册:[MiniMax](https://www.minimax.io/) -2. 获取API密钥 -3. 仪表板 → 添加 API 密钥 +1. Sign up: [Zhipu AI](https://open.bigmodel.cn/) +2. Get API key from Coding Plan +3. Dashboard → Add API Key: + - Provider: `glm` + - API Key: `your-key` -**使用:**`minimax/MiniMax-M2.1` +**Use:** `glm/glm-4.7` -**专业提示:**长上下文的最便宜选择(1M 代币)!### Kimi K2 ($9/month flat) +**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM. -1.订阅:【Moonshot AI】(https://platform.moonshot.ai/) 2. 获取API密钥 3. 仪表板 → 添加 API 密钥 +### MiniMax M2.1 (5h reset, $0.20/1M) -**使用:**`kimi/kimi-latest` +1. Sign up: [MiniMax](https://www.minimax.io/) +2. Get API key +3. Dashboard → Add API Key -**专业提示:**1000 万个代币固定为 9 美元/月 = 0.90 美元/100 万个有效成本!
+**Use:** `minimax/MiniMax-M2.1` -<详情> +**Pro Tip:** Cheapest option for long context (1M tokens)! -🆓 免费提供商(紧急备份)### Qoder (5 FREE models via OAuth) +### Kimi K2 ($9/month flat) + +1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/) +2. Get API key +3. Dashboard → Add API Key + +**Use:** `kimi/kimi-latest` + +**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost! + + + +
+🆓 FREE Providers (Emergency Backup) + +### Qoder (5 FREE models via OAuth) ```bash Dashboard → Connect Qoder @@ -1566,9 +1787,10 @@ Models:
-<详情> +
+🎨 Create Combos -🎨 创建组合### Example 1: Maximize Subscription → Cheap Backup +### Example 1: Maximize Subscription → Cheap Backup ``` Dashboard → Combos → Create New @@ -1596,9 +1818,10 @@ Cost: $0 forever!
-<详情> +
+🔧 CLI Integration -🔧 CLI 集成### Cursor IDE +### Cursor IDE ``` Settings → Models → Advanced: @@ -1609,7 +1832,9 @@ Settings → Models → Advanced: ### Claude Code -使用仪表板中的**CLI Tools**页面进行一键配置,或手动编辑 `~/.claude/settings.json`。### Codex CLI +Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually. + +### Codex CLI ```bash export OPENAI_BASE_URL="http://localhost:20128" @@ -1620,12 +1845,15 @@ codex "your prompt" ### OpenClaw -**选项 1 — 仪表板(推荐):**``` +**Option 1 — Dashboard (recommended):** + +``` Dashboard → CLI Tools → OpenClaw → Select Model → Apply +``` -```` +**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`: -**选项 2 — 手动:**编辑 `~/.openclaw/openclaw.json`:```json +```json { "models": { "providers": { @@ -1637,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply } } } -```` +``` -> **注意:**OpenClaw 仅适用于本地 OmniRoute。使用 `127.0.0.1` 而不是 `localhost` 以避免 IPv6 解析问题。### Cline / Continue / RooCode +> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues. + +### Cline / Continue / RooCode ``` Settings → API Configuration: @@ -1651,15 +1881,17 @@ Settings → API Configuration: ### OpenCode -**第 1 步:**添加 OmniRoute 作为自定义提供程序:```bash +**Step 1:** Add OmniRoute as a custom provider: + +```bash opencode /connect - # Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key +``` -```` +**Step 2:** Create/edit `opencode.json` in your project root: -**第 2 步:**在项目根目录中创建/编辑 `opencode.json`:```json +```json { "$schema": "https://opencode.ai/config.json", "provider": { @@ -1677,117 +1909,130 @@ opencode } } } -```` +``` -**步骤3:**在OpenCode中选择模型:```bash +**Step 3:** Select the model in OpenCode: + +```bash /models - # Select any OmniRoute model from the list +``` -```` +> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard. ->**提示:**将 OmniRoute `/v1/models` 端点中可用的任何模型添加到 `models` 部分。使用 OmniRoute 仪表板中的“provider/model-id”格式。
+ --- ## 故障排除 -<详情> -单击展开故障排除指南 +
+Click to expand troubleshooting guide -**“语言模型未提供消息”** +**"Language model did not provide messages"** -- 提供商配额耗尽 → 检查仪表板配额跟踪器 -- 解决方案:使用组合回退或切换到更便宜的层 +- Provider quota exhausted → Check dashboard quota tracker +- Solution: Use combo fallback or switch to cheaper tier -**速率限制** +**Rate limiting** -- 订阅配额耗尽 → 回退到 GLM/MiniMax -- 添加组合:`cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Subscription quota out → Fallback to GLM/MiniMax +- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -**OAuth 令牌已过期** +**OAuth token expired** -- 由 OmniRoute 自动刷新 -- 如果问题仍然存在:仪表板 → 提供商 → 重新连接 +- Auto-refreshed by OmniRoute +- If issues persist: Dashboard → Provider → Reconnect -**成本高** +**High costs** -- 在仪表板 → 成本中检查使用情况统计数据 -- 将主要模型切换为 GLM/MiniMax -- 使用免费层(Gemini CLI、Qoder)执行非关键任务 +- Check usage stats in Dashboard → Costs +- Switch primary model to GLM/MiniMax +- Use free tier (Gemini CLI, Qoder) for non-critical tasks -**仪表板/API端口错误** +**Dashboard/API ports are wrong** -- `PORT` 是规范的基本端口(默认情况下是 API 端口) -- `API_PORT` 仅覆盖 OpenAI 兼容的 API 监听器 -- `DASHBOARD_PORT` 仅覆盖仪表板/Next.js 监听器 -- 将“NEXT_PUBLIC_BASE_URL”设置为您的仪表板/公共 URL(用于 OAuth 回调) +- `PORT` is the canonical base port (and API port by default) +- `API_PORT` overrides only OpenAI-compatible API listener +- `DASHBOARD_PORT` overrides only dashboard/Next.js listener +- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks) -**云同步错误** +**Cloud sync errors** -- 验证“BASE_URL”指向您正在运行的实例 -- 验证“CLOUD_URL”指向您预期的云端点 -- 保持“NEXT_PUBLIC_*”值与服务器端值一致 +- Verify `BASE_URL` points to your running instance +- Verify `CLOUD_URL` points to your expected cloud endpoint +- Keep `NEXT_PUBLIC_*` values aligned with server-side values -**首次登录无法使用** +**First login not working** -- 检查“.env”中的“INITIAL_PASSWORD” -- 如果未设置,后备密码为“123456” +- Check `INITIAL_PASSWORD` in `.env` +- If unset, fallback password is `123456` -**没有请求日志** +**No request logs** -- 请求工件作为每个请求的一个 JSON 文件写入“DATA_DIR/call_logs/” -- 如果您需要详细的每阶段有效负载,请从仪表板 → 日志 → 请求日志启用管道捕获 -- 如果您还希望应用程序控制台日志记录在“logs/application/app.log”中,请设置“APP_LOG_TO_FILE=true” -- 根据需要调整`APP_LOG_MAX_FILE_SIZE`、`APP_LOG_RETENTION_DAYS`、`APP_LOG_MAX_FILES`和`CALL_LOG_MAX_ENTRIES` +- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request +- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads +- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log` +- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed -**连接测试显示 OpenAI 兼容提供商“无效”** +**Connection test shows "Invalid" for OpenAI-compatible providers** -- 许多提供商不公开“/models”端点 -- OmniRoute v1.0.6+ 包括通过聊天完成进行后备验证 -- 确保基本 URL 包含“/v1”后缀### 🔐 OAuth on a Remote Server +- Many providers don't expose a `/models` endpoint +- OmniRoute v1.0.6+ includes fallback validation via chat completions +- Ensure base URL includes `/v1` suffix + +### 🔐 OAuth on a Remote Server ->**⚠️ 对于在 VPS、Docker 或任何远程服务器上运行 OmniRoute 的用户来说很重要**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? +> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server** -**Antigravity**和**Gemini CLI**提供商使用**Google OAuth 2.0**。 Google 要求 OAuth 流程中的“redirect_uri”与应用的 Google Cloud Console 中预先注册的 URI 之一完全匹配。 +#### Why does Antigravity / Gemini CLI OAuth fail on remote servers? -OmniRoute 中捆绑的 OAuth 凭据仅针对“localhost”注册**。当您访问远程服务器上的 OmniRoute(例如“https://omniroute.myserver.com”)时,Google 将拒绝身份验证:``` +The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console. + +The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solution: Configure your own OAuth credentials -您需要使用服务器的 URI 在 Google Cloud Console 中创建**OAuth 2.0 客户端 ID**。#### Step-by-step +You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI. -**1.打开 Google Cloud Console** +#### Step-by-step -转到:[https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Open Google Cloud Console** -**2.创建新的 OAuth 2.0 客户端 ID** +Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- 单击**“+ 创建凭据”**→**“OAuth 客户端 ID”** -- 应用程序类型:**“Web 应用程序”** -- 名称:任何您喜欢的名称(例如“OmniRoute Remote”) +**2. Create a new OAuth 2.0 Client ID** -**3.添加授权重定向 URI** +- Click **"+ Create Credentials"** → **"OAuth client ID"** +- Application type: **"Web application"** +- Name: anything you like (e.g. `OmniRoute Remote`) -在**“授权重定向 URI”**字段中,添加:``` +**3. Add Authorized Redirect URIs** + +In the **"Authorized redirect URIs"** field, add: + +``` https://your-server.com/callback +``` -```` +> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`). -> 将 `your-server.com` 替换为您的服务器的域或 IP(如果需要,请包括端口,例如 `http://45.33.32.156:20128/callback`)。 +**4. Save and copy the credentials** -**4.保存并复制凭据** +After creating, Google will show the **Client ID** and **Client Secret**. -创建后,Google 将显示**Client ID**和**Client Secret**。 +**5. Set environment variables** -**5.设置环境变量** +In your `.env` (or Docker environment variables): -在你的 `.env` (或 Docker 环境变量)中:```bash +```bash # For Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret @@ -1796,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret -```` +``` -**6。重新启动 OmniRoute**```bash +**6. Restart OmniRoute** +```bash # npm: - npm run dev # Docker: - docker restart omniroute +``` -```` +**7. Try connecting again** -**7.尝试重新连接** +Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth -仪表板 → 提供商 → 反重力(或 Gemini CLI) → OAuth +Google will now redirect correctly to `https://your-server.com/callback`. -Google 现在将正确重定向到“https://your-server.com/callback”。--- +--- #### Temporary workaround (without custom credentials) -如果您现在不想设置自己的凭据,您仍然可以使用**手动 URL 流程**: +If you don't want to set up your own credentials right now, you can still use the **manual URL flow**: -1.OmniRoute打开Google授权URL -2.授权后,Google尝试重定向到`localhost`(在远程服务器上失败) -3.**从浏览器的地址栏中复制完整的 URL**(即使页面未加载) -4. 将该 URL 粘贴到 OmniRoute 连接模式中显示的字段中 -5. 单击**“连接”** +1. OmniRoute opens the Google authorization URL +2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server) +3. **Copy the full URL** from your browser's address bar (even if the page doesn't load) +4. Paste that URL into the field shown in the OmniRoute connection modal +5. Click **"Connect"** -> 这是有效的,因为无论是否加载重定向页面,URL 中的授权代码都是有效的。--- +> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded. -<详情> -🇧🇷 葡萄牙语版本#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? +--- -验证**反重力**和**Gemini CLI**使用**Google OAuth 2.0**进行验证。 O Google exige que a `redirect_uri` usada no Fluxo OAuth seja**exatamente**uma das URIs pre-cadastradas no Google Cloud Console do applicativo. +
+🇧🇷 Versão em Português -作为凭证,OAuth 不支持 OmniRoute estão cadastradas**apenas para `localhost`**。您可以通过远程服务访问 OmniRoute(例如:`https://omniroute.meuservidor.com`),或通过 Google 访问 autenticação com:``` +#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos? + +Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo. + +As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com: + +``` Error 400: redirect_uri_mismatch -```` +``` #### Solução: Configure suas próprias credenciais OAuth -请注意,**OAuth 2.0 客户端 ID**不是 Google Cloud Console com,而是 seu 服务器的 URI。#### Passo a passo +Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor. -**1.访问 Google Cloud Console** +#### Passo a passo -阿布拉:[https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) +**1. Acesse o Google Cloud Console** -**2.呐喊 OAuth 2.0 客户端 ID** +Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials) -- 派系**“+ 创建凭证”**→**“OAuth 客户端 ID”** -- 应用类型:**“Web 应用程序”** -- 名称:escolha qualquer nome(例如:`OmniRoute Remote`) +**2. Crie um novo OAuth 2.0 Client ID** -**3. Adicione 作为授权重定向 URI** +- Clique em **"+ Create Credentials"** → **"OAuth client ID"** +- Tipo de aplicativo: **"Web application"** +- Nome: escolha qualquer nome (ex: `OmniRoute Remote`) -没有坎波**“授权重定向 URI”**,adicione:``` +**3. Adicione as Authorized Redirect URIs** + +No campo **"Authorized redirect URIs"**, adicione: + +``` https://seu-servidor.com/callback +``` -```` +> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`). -> 将 `seu-servidor.com` 替换为 seu 服务的 IP 地址(包括必要的端口,例如:`http://45.33.32.156:20128/callback`)。 +**4. Salve e copie as credenciais** -**4.保存电子副本作为凭据** +Após criar, o Google mostrará o **Client ID** e o **Client Secret**. -请通过 Google 查询**客户端 ID**和**客户端秘密**。 +**5. Configure as variáveis de ambiente** -**5.配置为环境变量** +No seu `.env` (ou nas variáveis de ambiente do Docker): -没有 seu `.env`(或 Docker 环境变量):```bash +```bash # Para Antigravity: ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret @@ -1875,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret -```` +``` -**6。 Reinicie 或 OmniRoute**```bash +**6. Reinicie o OmniRoute** +```bash # Se usando npm: - npm run dev # Se usando Docker: - docker restart omniroute +``` -```` +**7. Tente conectar novamente** -**7.新连接的帐篷** +Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth -仪表板 → 提供商 → 反重力(ou Gemini CLI) → OAuth +Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará. -Agora 或 Google 重定向“https://seu-servidor.com/callback”和验证功能。--- +--- #### Workaround temporário (sem configurar credenciais próprias) -请参阅我们的官方文档,了解如何使用 Fluxo**URL 手册**: +Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**: -1. OmniRoute 是 Google 授权的 URL -2. 点击“localhost”,通过 Google 重新定向(que falha no server remoto) -3.**复制 URL 完整**da barra de endereço do seu browser (mesmo que a página não carregue) -4. 可以使用 OmniRoute 连接模式中的 URL -5. 拉帮结**“连接”** +1. O OmniRoute abrirá a URL de autorização do Google +2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto) +3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue) +4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute +5. Clique em **"Connect"** -> 这是一个解决方法,可以通过 URL 的自动控制功能来进行重定向,但不可以。
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não. + +
--- @@ -1913,64 +2171,73 @@ Agora 或 Google 重定向“https://seu-servidor.com/callback”和验证功能 ## 🛠️ Tech Stack -<详情> -点击展开技术堆栈详细信息 +
+Click to expand tech stack details --**运行时**:Node.js 18–22 LTS(⚠️ Node.js 24+**不支持**- `better-sqlite3` 本机二进制文件不兼容) --**语言**:TypeScript 5.9 —**跨 `src/` 和 `open-sse/` 的 100% TypeScript**(自 v2.0 以来核心模块中的“any”为零) --**框架**:Next.js 16 + React 19 + Tailwind CSS 4 --**数据库**:LowDB (JSON) + SQLite(域状态 + 代理日志 + MCP 审核 + 路由决策) --**模式**:Zod(MCP 工具 I/O 验证、API 合约) --**协议**:MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) --**流式传输**:服务器发送的事件 (SSE) --**Auth**:OAuth 2.0 (PKCE) + JWT + API 密钥 + MCP 范围授权 --**测试**:Node.js 测试运行器 + Vitest(900 多个测试,包括单元、集成、E2E) --**CI/CD**:GitHub Actions(自动 npm 发布 + Docker Hub 发布) --**网站**:[omniroute.online](https://omniroute.online) --**包**:[npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) --**Docker**:[hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) --**弹性**:断路器、指数退避、防雷群、TLS 欺骗、自动组合自愈
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible) +- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0) +- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4 +- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions) +- **Schemas**: Zod (MCP tool I/O validation, API contracts) +- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE) +- **Streaming**: Server-Sent Events (SSE) +- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization +- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E) +- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release) +- **Website**: [omniroute.online](https://omniroute.online) +- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute) +- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute) +- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing + + --- ## 文档 -|文件|描述 | -| ---------------------------------------------------------- | --------------------------------------------------- | -| [用户指南](docs/USER_GUIDE.md) |提供程序、组合、CLI 集成、部署 | -| [API参考](docs/API_REFERENCE.md) |所有端点及示例 | -| [MCP 服务器](open-sse/mcp-server/README.md) | 16 个 MCP 工具、IDE 配置、Python/TS/Go 客户端 | -| [A2A 服务器](src/lib/a2a/README.md) | JSON-RPC 2.0 协议、技能、流媒体、任务管理 | -| [自动组合引擎](docs/auto-combo.md) | 6 因素评分、模式包、自我修复 | -| [疑难解答](docs/TROUBLESHOOTING.md) |常见问题及解决办法 | -| [架构](docs/ARCHITECTURE.md) |系统架构和内部结构| -| [贡献](CONTRIBUTING.md) |开发设置和指南 | -| [OpenAPI 规范](docs/openapi.yaml) | OpenAPI 3.0 规范 | -| [安全策略](SECURITY.md) |漏洞报告和安全实践| -| [虚拟机部署](docs/VM_DEPLOYMENT_GUIDE.md) |完整指南:VM + nginx + Cloudflare 设置 | -| [功能图库](docs/FEATURES.md) |带有屏幕截图的可视化仪表板导览 | -| [发布清单](docs/RELEASE_CHECKLIST.md) |预发布验证步骤 |--- +| Document | Description | +| ---------------------------------------------- | --------------------------------------------------- | +| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment | +| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples | +| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients | +| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt | +| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing | +| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation | +| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions | +| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals | +| [Contributing](CONTRIBUTING.md) | Development setup and guidelines | +| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification | +| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices | +| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup | +| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots | +| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps | + +--- ## 🗺️ Roadmap -OmniRoute 在多个开发阶段规划了**210 多项功能**。以下是关键领域: +OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas: -|类别 |计划的功能|亮点| -| -------------------------------------- | ---------------- | ------------------------------------------------------------------------------------------ | -| 🧠**路由和智能**| 25+ |最低延迟路由、基于标签的路由、配额预检、P2C 账户选择 | -| 🔒**安全与合规性**| 20+ | SSRF 强化、凭证隐藏、每个端点的速率限制、管理密钥范围 | -| 📊**可观察性**| 15+ | OpenTelemetry 集成、实时配额监控、每个模型的成本跟踪 | -| 🔄**提供商集成**| 20+ |动态模型注册表、提供商冷却时间、多帐户 Codex、Copilot 配额解析 | -| ⚡**性能**| 15+ |双缓存层、提示缓存、响应缓存、流式保活、批处理 API | -| 🌐**生态系统**| 10+ | WebSocket API、配置热重载、分布式配置存储、商业模式 |### 🔜 Coming Soon +| Category | Planned Features | Highlights | +| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- | +| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection | +| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping | +| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model | +| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing | +| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API | +| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode | -- 🔗**OpenCode 集成**— 对 OpenCode AI 编码 IDE 的本机提供商支持 -- 🔗**TRAE 集成**— 全面支持 TRAE AI 开发框架 -- 📦**Batch API**— 批量请求的异步批处理 -- 🎯**基于标签的路由**— 基于自定义标签和元数据路由请求 -- 💰**最低成本策略**— 自动选择最便宜的可用提供商 +### 🔜 Coming Soon -> 📝 完整的功能规格可在 [`docs/new-features/`](docs/new-features/) 中找到(217 个详细规格)--- +- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE +- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework +- 📦 **Batch API** — Asynchronous batch processing for bulk requests +- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata +- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider + +> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs) + +--- ## 👥 Contributors @@ -1978,18 +2245,20 @@ OmniRoute 在多个开发阶段规划了**210 多项功能**。以下是关键 ### How to Contribute -1. 分叉存储库 -2. 创建您的功能分支(`git checkout -b feature/amazing-feature`) -3. 提交您的更改(`git commit -m '添加惊人的功能'`) -4.推送到分支(`git push origin feature/amazing-feature`) -5. 发起拉取请求 +1. Fork the repository +2. Create your feature branch (`git checkout -b feature/amazing-feature`) +3. Commit your changes (`git commit -m 'Add amazing feature'`) +4. Push to the branch (`git push origin feature/amazing-feature`) +5. Open a Pull Request -请参阅 [CONTRIBUTING.md](CONTRIBUTING.md) 了解详细指南。### Releasing a New Version +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines. + +### Releasing a New Version ```bash # Create a release — npm publish happens automatically gh release create v2.0.0 --title "v2.0.0" --generate-notes -```` +``` --- @@ -2001,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes ## 🙏 Acknowledgments -特别感谢**[decolua](https://github.com/decolua)**的**[9router](https://github.com/decolua/9router)**— 激发此分支的原始项目。 OmniRoute 建立在这个令人难以置信的基础上,具有附加功能、多模式 API 和完整的 TypeScript 重写。 +Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite. -特别感谢**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— 启发此 JavaScript 移植的原始 Go 实现。--- +Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port. + +--- ## 许可证 -MIT 许可证 - 有关详细信息,请参阅[许可证](许可证)。--- +MIT License - see [LICENSE](LICENSE) for details. + +---
Built with ❤️ for developers who code 24/7 diff --git a/docs/i18n/zh-CN/docs/ARCHITECTURE.md b/docs/i18n/zh-CN/docs/ARCHITECTURE.md index 9054534e50..ae37a17758 100644 --- a/docs/i18n/zh-CN/docs/ARCHITECTURE.md +++ b/docs/i18n/zh-CN/docs/ARCHITECTURE.md @@ -4,80 +4,93 @@ --- -_最后更新:2026-03-28_## Executive Summary -OmniRoute 是基于 Next.js 构建的本地 AI 路由网关和仪表板。 -它提供单个 OpenAI 兼容端点 (`/v1/*`),并通过转换、回退、令牌刷新和使用跟踪在多个上游提供商之间路由流量。 -核心能力: +_Last updated: 2026-03-28_ -- 用于 CLI/工具的 OpenAI 兼容 API 界面(28 个提供商) -- 跨提供商格式的请求/响应翻译 -- 模型组合后备(多模型序列) -- 账户级回退(每个提供商多个账户) -- OAuth + API 密钥提供商连接管理 -- 通过 `/v1/embeddings` 生成嵌入(6 个提供程序,9 个模型) -- 通过 `/v1/images/ Generations` 生成图像(4 个提供商,9 个模型) -- 用于推理模型的 Think 标签解析(`...`) -- 响应清理以实现严格的 OpenAI SDK 兼容性 -- 角色标准化(开发人员→系统、系统→用户)以实现跨提供商兼容性 -- 结构化输出转换(json_schema→Gemini responseSchema) -- 提供商、密钥、别名、组合、设置、定价的本地持久性 -- 使用/成本跟踪和请求记录 -- 可选的云同步用于多设备/状态同步 -- API 访问控制的 IP 允许列表/阻止列表 -- 思考预算管理(直通/自动/自定义/自适应) -- 全局系统提示注入 -- 会话跟踪和指纹识别 -- 使用特定于提供商的配置文件增强每个帐户的速率限制 -- 提供者弹性的断路器模式 -- 具有互斥锁的防雷群保护 -- 基于签名的请求重复数据删除缓存 -- 领域层:模型可用性、成本规则、后备策略、锁定策略 -- 域状态持久性(用于回退、预算、锁定、断路器的 SQLite 直写式缓存) -- 用于集中请求评估的策略引擎(锁定→预算→后备) -- 使用 p50/p95/p99 延迟聚合请求遥测 -- 用于端到端跟踪的关联 ID (X-Request-Id) -- 合规性审核日志记录,可根据 API 密钥选择退出 -- LLM质量保证评估框架 -- 具有实时断路器状态的 Resilience UI 仪表板 -- 模块化 OAuth 提供程序(`src/lib/oauth/providers/` 下有 12 个单独的模块) +## Executive Summary -主要运行时模型: +OmniRoute is a local AI routing gateway and dashboard built on Next.js. +It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking. -- `src/app/api/*` 下的 Next.js 应用程序路由实现了仪表板 API 和兼容性 API -- `src/sse/*` + `open-sse/*` 中的共享 SSE/路由核心处理提供程序执行、转换、流式传输、回退和使用## Scope and Boundaries +Core capabilities: + +- OpenAI-compatible API surface for CLI/tools (28 providers) +- Request/response translation across provider formats +- Model combo fallback (multi-model sequence) +- Account-level fallback (multi-account per provider) +- OAuth + API-key provider connection management +- Embedding generation via `/v1/embeddings` (6 providers, 9 models) +- Image generation via `/v1/images/generations` (4 providers, 9 models) +- Think tag parsing (`...`) for reasoning models +- Response sanitization for strict OpenAI SDK compatibility +- Role normalization (developer→system, system→user) for cross-provider compatibility +- Structured output conversion (json_schema → Gemini responseSchema) +- Local persistence for providers, keys, aliases, combos, settings, pricing +- Usage/cost tracking and request logging +- Optional cloud sync for multi-device/state sync +- IP allowlist/blocklist for API access control +- Thinking budget management (passthrough/auto/custom/adaptive) +- Global system prompt injection +- Session tracking and fingerprinting +- Per-account enhanced rate limiting with provider-specific profiles +- Circuit breaker pattern for provider resilience +- Anti-thundering herd protection with mutex locking +- Signature-based request deduplication cache +- Domain layer: model availability, cost rules, fallback policy, lockout policy +- Context Relay: session handoff summaries for account rotation continuity +- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers) +- Policy engine for centralized request evaluation (lockout → budget → fallback) +- Request telemetry with p50/p95/p99 latency aggregation +- Correlation ID (X-Request-Id) for end-to-end tracing +- Compliance audit logging with opt-out per API key +- Eval framework for LLM quality assurance +- Resilience UI dashboard with real-time circuit breaker status +- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`) + +Primary runtime model: + +- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs +- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage + +## Scope and Boundaries ### In Scope -- 本地网关运行时 -- 仪表板管理 API -- 提供商身份验证和令牌刷新 -- 请求翻译和 SSE 流媒体 -- 本地状态+使用持久性 -- 可选的云同步编排### Out of Scope +- Local gateway runtime +- Dashboard management APIs +- Provider authentication and token refresh +- Request translation and SSE streaming +- Local state + usage persistence +- Optional cloud sync orchestration -- “NEXT_PUBLIC_CLOUD_URL”背后的云服务实现 -- 本地流程之外的提供商 SLA/控制平面 -- 外部 CLI 二进制文件本身(Claude CLI、Codex CLI 等)## Dashboard Surface (Current) +### Out of Scope -`src/app/(dashboard)/dashboard/`下的主要页面: +- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL` +- Provider SLA/control plane outside local process +- External CLI binaries themselves (Claude CLI, Codex CLI, etc.) -- `/dashboard` — 快速启动 + 提供商概述 -- `/dashboard/endpoint` — 端点代理 + MCP + A2A + API 端点选项卡 -- `/dashboard/providers` — 提供商连接和凭证 -- `/dashboard/combos` — 组合策略、模板、模型路由规则 -- `/dashboard/costs` — 成本汇总和定价可见性 -- `/dashboard/analytics` — 使用情况分析和评估 -- `/dashboard/limits` — 配额/速率控制 -- `/dashboard/cli-tools` — CLI 入门、运行时检测、配置生成 -- `/dashboard/agents` — 检测到的 ACP 代理 + 自定义代理注册 -- `/dashboard/media` — 图像/视频/音乐游乐场 -- `/dashboard/search-tools` — 搜索提供商测试和历史记录 -- `/dashboard/health` — 正常运行时间、断路器、速率限制 -- `/dashboard/logs` — 请求/代理/审核/控制台日志 -- `/dashboard/settings` — 系统设置选项卡(常规、路由、组合默认值等) -- `/dashboard/api-manager` — API 密钥生命周期和模型权限## High-Level System Context +## Dashboard Surface (Current) + +Main pages under `src/app/(dashboard)/dashboard/`: + +- `/dashboard` — quick start + provider overview +- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs +- `/dashboard/providers` — provider connections and credentials +- `/dashboard/combos` — combo strategies, templates, model routing rules +- `/dashboard/costs` — cost aggregation and pricing visibility +- `/dashboard/analytics` — usage analytics and evaluations +- `/dashboard/limits` — quota/rate controls +- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation +- `/dashboard/agents` — detected ACP agents + custom agent registration +- `/dashboard/media` — image/video/music playground +- `/dashboard/search-tools` — search provider testing and history +- `/dashboard/health` — uptime, circuit breakers, rate limits +- `/dashboard/logs` — request/proxy/audit/console logs +- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.) +- `/dashboard/api-manager` — API key lifecycle and model permissions + +## High-Level System Context ```mermaid flowchart LR @@ -129,139 +142,151 @@ flowchart LR ## 1) API and Routing Layer (Next.js App Routes) -主要目录: +Main directories: -- `src/app/api/v1/*` 和 `src/app/api/v1beta/*` 用于兼容性 API -- `src/app/api/*` 用于管理/配置 API -- Next 在 `next.config.mjs` 中重写,将 `/v1/*` 映射到 `/api/v1/*` +- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs +- `src/app/api/*` for management/configuration APIs +- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*` -重要的兼容性路线: +Important compatibility routes: - `src/app/api/v1/chat/completions/route.ts` - `src/app/api/v1/messages/route.ts` - `src/app/api/v1/responses/route.ts` -- `src/app/api/v1/models/route.ts` — 包括带有 `custom: true` 的自定义模型 -- `src/app/api/v1/embeddings/route.ts` — 嵌入生成(6 个提供程序) -- `src/app/api/v1/images/ Generations/route.ts` — 图像生成(4 个以上提供商,包括 Antigravity/Nebius) +- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true` +- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers) +- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius) - `src/app/api/v1/messages/count_tokens/route.ts` -- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — 每个提供商的专用聊天 -- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — 专用的每个提供商嵌入 -- `src/app/api/v1/providers/[provider]/images/ Generations/route.ts` — 每个提供商专用的图像 +- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat +- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings +- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images - `src/app/api/v1beta/models/route.ts` - `src/app/api/v1beta/models/[...path]/route.ts` -管理域: +Management domains: -- 身份验证/设置:`src/app/api/auth/*`、`src/app/api/settings/*` -- 提供者/连接:`src/app/api/providers*` -- 提供者节点:`src/app/api/provider-nodes*` -- 自定义模型:`src/app/api/provider-models` (GET/POST/DELETE) -- 模型目录:`src/app/api/models/route.ts` (GET) -- 代理配置:`src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) -- OAuth:`src/app/api/oauth/*` -- 键/别名/组合/定价:`src/app/api/keys*`、`src/app/api/models/alias`、`src/app/api/combos*`、`src/app/api/pricing` -- 用法:`src/app/api/usage/*` -- 同步/云:`src/app/api/sync/*`、`src/app/api/cloud/*` -- CLI 工具助手:`src/app/api/cli-tools/*` -- IP 过滤器:`src/app/api/settings/ip-filter` (GET/PUT) -- 思考预算:`src/app/api/settings/thinking-budget` (GET/PUT) -- 系统提示:`src/app/api/settings/system-prompt` (GET/PUT) -- 会话:`src/app/api/sessions` (GET) -- 速率限制:`src/app/api/rate-limits` (GET) -- 弹性:`src/app/api/resilience` (GET/PATCH) — 提供商配置文件、断路器、速率限制状态 -- 弹性重置:`src/app/api/resilience/reset` (POST) — 重置断路器 + 冷却时间 -- 缓存统计信息:`src/app/api/cache/stats`(获取/删除) -- 模型可用性:`src/app/api/models/availability` (GET/POST) -- 遥测:`src/app/api/telemetry/summary` (GET) -- 预算:`src/app/api/usage/budget` (GET/POST) -- 后备链:`src/app/api/fallback/chains` (GET/POST/DELETE) -- 合规性审计:`src/app/api/compliance/audit-log` (GET) -- 评估:`src/app/api/evals` (GET/POST)、`src/app/api/evals/[suiteId]` (GET) -- 政策:`src/app/api/policies` (GET/POST)## 2) SSE + Translation Core +- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*` +- Providers/connections: `src/app/api/providers*` +- Provider nodes: `src/app/api/provider-nodes*` +- Custom models: `src/app/api/provider-models` (GET/POST/DELETE) +- Model catalog: `src/app/api/models/route.ts` (GET) +- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST) +- OAuth: `src/app/api/oauth/*` +- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing` +- Usage: `src/app/api/usage/*` +- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*` +- CLI tooling helpers: `src/app/api/cli-tools/*` +- IP filter: `src/app/api/settings/ip-filter` (GET/PUT) +- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT) +- System prompt: `src/app/api/settings/system-prompt` (GET/PUT) +- Sessions: `src/app/api/sessions` (GET) +- Rate limits: `src/app/api/rate-limits` (GET) +- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state +- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns +- Cache stats: `src/app/api/cache/stats` (GET/DELETE) +- Model availability: `src/app/api/models/availability` (GET/POST) +- Telemetry: `src/app/api/telemetry/summary` (GET) +- Budget: `src/app/api/usage/budget` (GET/POST) +- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE) +- Compliance audit: `src/app/api/compliance/audit-log` (GET) +- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET) +- Policies: `src/app/api/policies` (GET/POST) -主要流程模块: +## 2) SSE + Translation Core -- 条目:`src/sse/handlers/chat.ts` -- 核心编排:`open-sse/handlers/chatCore.ts` -- 提供程序执行适配器:`open-sse/executors/*` -- 格式检测/提供程序配置:`open-sse/services/provider.ts` -- 模型解析/解析:`src/sse/services/model.ts`、`open-sse/services/model.ts` -- 帐户后备逻辑:`open-sse/services/accountFallback.ts` -- 翻译注册表:`open-sse/translator/index.ts` -- 流转换:`open-sse/utils/stream.ts`、`open-sse/utils/streamHandler.ts` -- 使用情况提取/规范化:`open-sse/utils/usageTracking.ts` -- Think 标签解析器:`open-sse/utils/thinkTagParser.ts` -- 嵌入处理程序:`open-sse/handlers/embeddings.ts` -- 嵌入提供程序注册表:`open-sse/config/embeddingRegistry.ts` -- 图像生成处理程序:`open-sse/handlers/imageGeneration.ts` -- 图像提供程序注册表:`open-sse/config/imageRegistry.ts` -- 响应清理:`open-sse/handlers/responseSanitizer.ts` -- 角色规范化:`open-sse/services/roleNormalizer.ts` +Main flow modules: -服务(业务逻辑): +- Entry: `src/sse/handlers/chat.ts` +- Core orchestration: `open-sse/handlers/chatCore.ts` +- Provider execution adapters: `open-sse/executors/*` +- Format detection/provider config: `open-sse/services/provider.ts` +- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts` +- Account fallback logic: `open-sse/services/accountFallback.ts` +- Translation registry: `open-sse/translator/index.ts` +- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts` +- Usage extraction/normalization: `open-sse/utils/usageTracking.ts` +- Think tag parser: `open-sse/utils/thinkTagParser.ts` +- Embedding handler: `open-sse/handlers/embeddings.ts` +- Embedding provider registry: `open-sse/config/embeddingRegistry.ts` +- Image generation handler: `open-sse/handlers/imageGeneration.ts` +- Image provider registry: `open-sse/config/imageRegistry.ts` +- Response sanitization: `open-sse/handlers/responseSanitizer.ts` +- Role normalization: `open-sse/services/roleNormalizer.ts` -- 账户选择/评分:`open-sse/services/accountSelector.ts` -- 上下文生命周期管理:`open-sse/services/contextManager.ts` -- IP 过滤器强制执行:`open-sse/services/ipFilter.ts` -- 会话跟踪:`open-sse/services/sessionManager.ts` -- 请求重复数据删除:`open-sse/services/signatureCache.ts` -- 系统提示注入:`open-sse/services/systemPrompt.ts` -- 思维预算管理:`open-sse/services/thinkingBudget.ts` -- 通配符模型路由:`open-sse/services/wildcardRouter.ts` -- 速率限制管理:`open-sse/services/rateLimitManager.ts` -- 断路器:`open-sse/services/CircuitBreaker.ts` +Services (business logic): -领域层模块: +- Account selection/scoring: `open-sse/services/accountSelector.ts` +- Context lifecycle management: `open-sse/services/contextManager.ts` +- IP filter enforcement: `open-sse/services/ipFilter.ts` +- Session tracking: `open-sse/services/sessionManager.ts` +- Request deduplication: `open-sse/services/signatureCache.ts` +- System prompt injection: `open-sse/services/systemPrompt.ts` +- Thinking budget management: `open-sse/services/thinkingBudget.ts` +- Wildcard model routing: `open-sse/services/wildcardRouter.ts` +- Rate limit management: `open-sse/services/rateLimitManager.ts` +- Circuit breaker: `open-sse/services/circuitBreaker.ts` +- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy +- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions -- 模型可用性:`src/lib/domain/modelAvailability.ts` -- 成本规则/预算:`src/lib/domain/costRules.ts` -- 后备策略:`src/lib/domain/fallbackPolicy.ts` -- 组合解析器:`src/lib/domain/comboResolver.ts` -- 锁定策略:`src/lib/domain/lockoutPolicy.ts` -- 策略引擎:`src/domain/policyEngine.ts` — 集中锁定→预算→后备评估 -- 错误代码目录:`src/lib/domain/errorCodes.ts` -- 请求 ID:`src/lib/domain/requestId.ts` -- 获取超时:`src/lib/domain/fetchTimeout.ts` -- 请求遥测:`src/lib/domain/requestTelemetry.ts` -- 合规性/审核:`src/lib/domain/compliance/index.ts` -- 评估运行器:`src/lib/domain/evalRunner.ts` -- 域状态持久化:`src/lib/db/domainState.ts` — 用于后备链、预算、成本历史记录、锁定状态、断路器的 SQLite CRUD +Domain layer modules: -OAuth 提供程序模块(`src/lib/oauth/providers/` 下有 12 个单独的文件): +- Model availability: `src/lib/domain/modelAvailability.ts` +- Cost rules/budgets: `src/lib/domain/costRules.ts` +- Fallback policy: `src/lib/domain/fallbackPolicy.ts` +- Combo resolver: `src/lib/domain/comboResolver.ts` +- Lockout policy: `src/lib/domain/lockoutPolicy.ts` +- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation +- Error codes catalog: `src/lib/domain/errorCodes.ts` +- Request ID: `src/lib/domain/requestId.ts` +- Fetch timeout: `src/lib/domain/fetchTimeout.ts` +- Request telemetry: `src/lib/domain/requestTelemetry.ts` +- Compliance/audit: `src/lib/domain/compliance/index.ts` +- Eval runner: `src/lib/domain/evalRunner.ts` +- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers -- 注册表索引:`src/lib/oauth/providers/index.ts` -- 个别提供者:`claude.ts`、`codex.ts`、`gemini.ts`、`antigravity.ts`、`qoder.ts`、`qwen.ts`、`kimi-coding.ts`、`github.ts`、`kiro.ts`、`cursor.ts`、`kilocode.ts`、`cline.ts` -- 薄包装器:`src/lib/oauth/providers.ts` — 从各个模块重新导出## 3) Persistence Layer +OAuth provider modules (12 individual files under `src/lib/oauth/providers/`): -主状态数据库(SQLite): +- Registry index: `src/lib/oauth/providers/index.ts` +- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts` +- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules -- 核心基础设施:`src/lib/db/core.ts`(better-sqlite3、迁移、WAL) -- 重新导出外观:`src/lib/localDb.ts`(调用者的薄兼容层) -- 文件:`${DATA_DIR}/storage.sqlite`(或设置时为`$XDG_CONFIG_HOME/omniroute/storage.sqlite`,否则为`~/.omniroute/storage.sqlite`) -- 实体(表 + KV 命名空间):providerConnections、providerNodes、modelAliases、组合、apiKeys、设置、定价、**customModels**、**proxyConfig**、**ipFilter**、**thinkingBudget**、**systemPrompt** +## 3) Persistence Layer -使用持久性: +Primary state DB (SQLite): -- 门面:`src/lib/usageDb.ts`(在`src/lib/usage/*`中分解模块) -- `storage.sqlite` 中的 SQLite 表:`usage_history`、`call_logs`、`proxy_logs` -- 保留可选文件工件以实现兼容性/调试(`${DATA_DIR}/log.txt`、`${DATA_DIR}/call_logs/`、`/logs/...`) -- 旧版 JSON 文件通过启动迁移(如果存在)迁移到 SQLite +- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL) +- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers) +- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`) +- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt** -域状态数据库(SQLite): +Usage persistence: -- `src/lib/db/domainState.ts` — 域状态的 CRUD 操作 -- 表(在 `src/lib/db/core.ts` 中创建):`domain_fallback_chains`、`domain_budgets`、`domain_cost_history`、`domain_lockout_state`、`domain_Circuit_breakers` -- 直写式缓存模式:内存中的Map在运行时具有权威性;突变同步写入SQLite;冷启动时从数据库恢复状态## 4) Auth + Security Surfaces +- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`) +- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs` +- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`) +- legacy JSON files are migrated to SQLite by startup migrations when present -- 仪表板 cookie 身份验证:`src/proxy.ts`、`src/app/api/auth/login/route.ts` -- API 密钥生成/验证:`src/shared/utils/apiKey.ts` -- 提供商机密保留在“providerConnections”条目中 -- 通过“open-sse/utils/proxyFetch.ts”(环境变量)和“open-sse/utils/networkProxy.ts”(可按提供商或全局配置)提供出站代理支持## 5) Cloud Sync +Domain State DB (SQLite): -- 调度程序初始化:`src/lib/initCloudSync.ts`、`src/shared/services/initializeCloudSync.ts`、`src/shared/services/modelSyncScheduler.ts` -- 定期任务:`src/shared/services/cloudSyncScheduler.ts` -- 定期任务:`src/shared/services/modelSyncScheduler.ts` -- 控制路由:`src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`) +- `src/lib/db/domainState.ts` — CRUD operations for domain state +- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers` +- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start + +## 4) Auth + Security Surfaces + +- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts` +- API key generation/verification: `src/shared/utils/apiKey.ts` +- Provider secrets persisted in `providerConnections` entries +- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global) + +## 5) Cloud Sync + +- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts` +- Periodic task: `src/shared/services/cloudSyncScheduler.ts` +- Periodic task: `src/shared/services/modelSyncScheduler.ts` +- Control route: `src/app/api/sync/cloud/route.ts` + +## Request Lifecycle (`/v1/chat/completions`) ```mermaid sequenceDiagram @@ -338,7 +363,9 @@ flowchart TD Q -- No --> R[Return all unavailable] ``` -回退决策由“open-sse/services/accountFallback.ts”使用状态代码和错误消息启发法驱动。组合路由增加了一项额外的防护:提供者范围内的 400(例如上游内容块和角色验证失败)被视为模型本地失败,因此后面的组合目标仍然可以运行。## OAuth Onboarding and Token Refresh Lifecycle +Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run. + +## OAuth Onboarding and Token Refresh Lifecycle ```mermaid sequenceDiagram @@ -368,7 +395,9 @@ sequenceDiagram Test-->>UI: validation result ``` -实时流量期间的刷新通过执行器“refreshCredentials()”在“open-sse/handlers/chatCore.ts”内执行。## Cloud Sync Lifecycle (Enable / Sync / Disable) +Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`. + +## Cloud Sync Lifecycle (Enable / Sync / Disable) ```mermaid sequenceDiagram @@ -400,7 +429,9 @@ sequenceDiagram Sync-->>UI: disabled ``` -启用云时,定期同步由“CloudSyncScheduler”触发。## Data Model and Storage Map +Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled. + +## Data Model and Storage Map ```mermaid erDiagram @@ -501,12 +532,14 @@ erDiagram } ``` -物理存储文件: +Physical storage files: -- 主运行时数据库:`${DATA_DIR}/storage.sqlite` -- 请求日志行:`${DATA_DIR}/log.txt`(兼容/调试工件) -- 结构化调用有效负载档案:`${DATA_DIR}/call_logs/` -- 可选的转换器/请求调试会话:`/logs/...`## Deployment Topology +- primary runtime DB: `${DATA_DIR}/storage.sqlite` +- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact) +- structured call payload archives: `${DATA_DIR}/call_logs/` +- optional translator/request debug sessions: `/logs/...` + +## Deployment Topology ```mermaid flowchart LR @@ -541,204 +574,249 @@ flowchart LR ### Route and API Modules -- `src/app/api/v1/*`、`src/app/api/v1beta/*`:兼容性 API -- `src/app/api/v1/providers/[provider]/*`:每个提供商的专用路由(聊天、嵌入、图像) -- `src/app/api/providers*`:提供者 CRUD、验证、测试 -- `src/app/api/provider-nodes*`:自定义兼容节点管理 -- `src/app/api/provider-models`:自定义模型管理(CRUD) -- `src/app/api/models/route.ts`:模型目录 API(别名+自定义模型) -- `src/app/api/oauth/*`:OAuth/设备代码流 -- `src/app/api/keys*`:本地 API 密钥生命周期 -- `src/app/api/models/alias`:别名管理 -- `src/app/api/combos*`:后备组合管理 -- `src/app/api/pricing`:成本计算的定价覆盖 -- `src/app/api/settings/proxy`:代理配置(GET/PUT/DELETE)-`src/app/api/settings/proxy/test`:出站代理连接测试(POST) -- `src/app/api/usage/*`:使用情况和日志 API -- `src/app/api/sync/*` + `src/app/api/cloud/*`:云同步和面向云的助手 -- `src/app/api/cli-tools/*`:本地 CLI 配置编写器/检查器 -- `src/app/api/settings/ip-filter`: IP 允许列表/阻止列表 (GET/PUT) -- `src/app/api/settings/thinking-budget`:思考代币预算配置(GET/PUT) -- `src/app/api/settings/system-prompt`:全局系统提示符(GET/PUT) -- `src/app/api/sessions`:活动会话列表(GET) -- `src/app/api/rate-limits`:每个帐户的速率限制状态 (GET)### Routing and Execution Core +- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs +- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images) +- `src/app/api/providers*`: provider CRUD, validation, testing +- `src/app/api/provider-nodes*`: custom compatible node management +- `src/app/api/provider-models`: custom model management (CRUD) +- `src/app/api/models/route.ts`: model catalog API (aliases + custom models) +- `src/app/api/oauth/*`: OAuth/device-code flows +- `src/app/api/keys*`: local API key lifecycle +- `src/app/api/models/alias`: alias management +- `src/app/api/combos*`: fallback combo management +- `src/app/api/pricing`: pricing overrides for cost calculation +- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE) +- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST) +- `src/app/api/usage/*`: usage and logs APIs +- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers +- `src/app/api/cli-tools/*`: local CLI config writers/checkers +- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT) +- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT) +- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT) +- `src/app/api/sessions`: active session listing (GET) +- `src/app/api/rate-limits`: per-account rate limit status (GET) -- `src/sse/handlers/chat.ts`:请求解析、组合处理、帐户选择循环 -- `open-sse/handlers/chatCore.ts`:翻译、执行器调度、重试/刷新处理、流设置 -- `open-sse/executors/*`:特定于提供商的网络和格式行为### Translation Registry and Format Converters +### Routing and Execution Core -- `open-sse/translator/index.ts`:翻译器注册表和编排 -- 请求翻译器:`open-sse/translator/request/*` -- 响应翻译器:`open-sse/translator/response/*` -- 格式常量:`open-sse/translator/formats.ts`### Persistence +- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop +- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup +- `open-sse/executors/*`: provider-specific network and format behavior -- `src/lib/db/*`:SQLite 上的持久配置/状态和域持久性 -- `src/lib/localDb.ts`:数据库模块的兼容性重新导出 -- `src/lib/usageDb.ts`:SQLite 表顶部的使用历史记录/调用日志外观## Provider Executor Coverage (Strategy Pattern) +### Translation Registry and Format Converters -每个提供者都有一个扩展“BaseExecutor”的专门执行器(在“open-sse/executors/base.ts”中),它提供 URL 构建、标头构建、指数退避重试、凭证刷新挂钩和“execute()”编排方法。 +- `open-sse/translator/index.ts`: translator registry and orchestration +- Request translators: `open-sse/translator/request/*` +- Response translators: `open-sse/translator/response/*` +- Format constants: `open-sse/translator/formats.ts` -| 执行人 | 提供商 | 特殊处理 | -| ------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------ | ------------------------------------------------------------ | -| `默认执行器` | OpenAI、Claude、Gemini、Qwen、Qoder、OpenRouter、GLM、Kimi、MiniMax、DeepSeek、Groq、xAI、Mistral、Perplexity、Together、Fireworks、Cerebras、Cohere、NVIDIA | 每个提供商的动态 URL/标头配置 | -| `反重力执行者` | 谷歌反重力 | 自定义项目/会话 ID,解析后重试 | -| `CodexExecutor` | OpenAI 法典 | 注入系统指令,强制推理工作 | -| `CursorExecutor` | 光标IDE | ConnectRPC 协议、Protobuf 编码、通过校验和进行请求签名 | -| `GithubExecutor` | GitHub 副驾驶 | Copilot 令牌刷新,模仿 VSCode 标头 | -| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS CodeWhisperer/Kiro | AWS CodeWhisperer/Kiro AWS EventStream 二进制格式 → SSE 转换 | -| `GeminiCLIExecutor` | 双子座 CLI | Google OAuth 令牌刷新周期 | +### Persistence -所有其他提供者(包括自定义兼容节点)都使用“DefaultExecutor”。## Provider Compatibility Matrix +- `src/lib/db/*`: persistent config/state and domain persistence on SQLite +- `src/lib/localDb.ts`: compatibility re-export for DB modules +- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables -| 供应商 | 格式 | 授权 | 流 | 非流 | 令牌刷新 | 使用API​​ | -| ---------------- | ----------- | ------------------ | ------------ | ------------------------- | -------- | -------------- | ------------------------------ | -| 克劳德 | 克劳德 | API 密钥/OAuth | ✅ | ✅ | ✅ | ⚠️ 仅限管理员 | -| 双子座 | 双子座 | API 密钥/OAuth | ✅ | ✅ | ✅ | ⚠️ 云控制台 | -| 双子座 CLI | Gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ 云控制台 | -| 反重力 | 反重力 | OAuth | ✅ | ✅ | ✅ | ✅ 完整配额API | -| 开放人工智能 | 开放 | API 密钥 | ✅ | ✅ | ❌ | ❌ | -| 法典 | openai-回应 | OAuth | ✅ 强迫 | ❌ | ✅ | ✅ 速率限制 | -| GitHub 副驾驶 | 开放 | OAuth + 副驾驶令牌 | ✅ | ✅ | ✅ | ✅ 配额快照 | -| 光标 | 光标 | 自定义校验和 | ✅ | ✅ | ❌ | ❌ | -| 基罗 | 基罗 | AWS SSO OIDC | AWS SSO OIDC | AWS SSO OIDC ✅(事件流) | ❌ | ✅ | ✅ 使用限制 | -| 奎文 | 开放 | OAuth | ✅ | ✅ | ✅ | ⚠️ 根据要求 | -| 科德尔 | 开放 | OAuth(基本) | ✅ | ✅ | ✅ | ⚠️ 根据要求 | -| 开放路由器 | 开放 | API 密钥 | ✅ | ✅ | ❌ | ❌ | -| GLM/Kimi/MiniMax | 克劳德 | API 密钥 | ✅ | ✅ | ❌ | ❌ | -| 深度搜索 | 开放 | API 密钥 | ✅ | ✅ | ❌ | ❌ | -| 格罗克 | 开放 | API 密钥 | ✅ | ✅ | ❌ | ❌ | -| xAI (Grok) | 开放 | API 密钥 | ✅ | ✅ | ❌ | ❌ | -| 米斯特拉尔 | 开放 | API 密钥 | ✅ | ✅ | ❌ | ❌ | -| 困惑 | 开放 | API 密钥 | ✅ | ✅ | ❌ | ❌ | -| 一起人工智能 | 开放 | API 密钥 | ✅ | ✅ | ❌ | ❌ | -| 烟花人工智能 | 开放 | API 密钥 | ✅ | ✅ | ❌ | ❌ | -| 大脑 | 开放 | API 密钥 | ✅ | ✅ | ❌ | ❌ | -| 连贯 | 开放 | API 密钥 | ✅ | ✅ | ❌ | ❌ | -| NVIDIA NIM | 开放 | API 密钥 | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage | +## Provider Executor Coverage (Strategy Pattern) -检测到的源格式包括: +Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method. + +| Executor | Provider(s) | Special Handling | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | +| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider | +| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing | +| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort | +| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum | +| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers | +| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion | +| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle | + +All other providers (including custom compatible nodes) use the `DefaultExecutor`. + +## Provider Compatibility Matrix + +| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API | +| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ | +| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only | +| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console | +| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API | +| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits | +| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots | +| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ | +| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits | +| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request | +| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request | +| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ | +| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ | +| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ | + +## Format Translation Coverage + +Detected source formats include: - `openai` -- `openai-响应` - -“克劳德” - -“双子座” +- `openai-responses` +- `claude` +- `gemini` -目标格式包括: +Target formats include: -- OpenAI 聊天/回复 - ——克劳德 -- Gemini/Gemini-CLI/反重力信封 -- 基罗 -- 光标 +- OpenAI chat/Responses +- Claude +- Gemini/Gemini-CLI/Antigravity envelope +- Kiro +- Cursor -翻译使用**OpenAI 作为中心格式**- 所有转换都通过 OpenAI 作为中间:``` +Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate: + +``` Source Format → OpenAI (hub) → Target Format +``` -```` +Translations are selected dynamically based on source payload shape and provider target format. -根据源有效负载形状和提供程序目标格式动态选择翻译。 +Additional processing layers in the translation pipeline: -翻译管道中的附加处理层: +- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance +- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE) +- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field +- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema` --**响应清理**- 从 OpenAI 格式响应(流式和非流式)中去除非标准字段,以确保严格的 SDK 合规性 --**角色规范化**— 对于非 OpenAI 目标,将“开发人员”转换为“系统”;对于拒绝系统角色的模型(GLM、ERNIE),合并“system”→“user” --**Think 标签提取**— 将内容中的 `...` 块解析到 `reasoning_content` 字段中 --**结构化输出**— 将 OpenAI `response_format.json_schema` 转换为 Gemini 的 `responseMimeType` + `responseSchema`## Supported API Endpoints +## Supported API Endpoints -|端点 |格式|处理程序 | -| -------------------------------------------------- | ------------------ | ---------------------------------------------------------------------------------- | -| `POST /v1/chat/completions` | OpenAI 聊天 | `src/sse/handlers/chat.ts` | -| `POST /v1/messages` |克劳德消息 |相同的处理程序(自动检测)| -| `POST /v1/response` | OpenAI 回应 | `open-sse/handlers/responsesHandler.ts` | -| `POST /v1/embeddings` | OpenAI 嵌入 | `open-sse/handlers/embeddings.ts` | -| `GET /v1/embeddings` |型号列表 | API路线 | -| `POST /v1/images/Generations` | OpenAI 图像 | `open-sse/handlers/imageGeneration.ts` | -| `获取/v1/图像/世代` |型号列表 | API路线 | -| `POST /v1/providers/{provider}/chat/completions` | OpenAI 聊天 |专用于每个提供商的模型验证 | -| `POST /v1/providers/{provider}/embeddings` | OpenAI 嵌入 |专用于每个提供商的模型验证 | -| `POST /v1/providers/{provider}/images/ Generations` | OpenAI 图像 |专用于每个提供商的模型验证 | -| `POST /v1/messages/count_tokens` |克劳德代币计数 | API路线 | -| `获取/v1/模型` | OpenAI 模型列表 | API路线(聊天+嵌入+图像+自定义模型)| -| `GET /api/models/catalog` |目录|所有模型按提供商+类型分组 | -| `POST /v1beta/models/*:streamGenerateContent` |双子座人 | API路线| -| `获取/放置/删除 /api/settings/proxy` |代理配置 |网络代理配置| -| `POST /api/settings/proxy/test` |代理连接 |代理运行状况/连接测试端点 | -| `GET/POST/DELETE /api/provider-models` |供应商模型|支持自定义和托管可用模型的提供者模型元数据 |## Bypass Handler +| Endpoint | Format | Handler | +| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- | +| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` | +| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) | +| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` | +| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` | +| `GET /v1/embeddings` | Model listing | API route | +| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` | +| `GET /v1/images/generations` | Model listing | API route | +| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation | +| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation | +| `POST /v1/messages/count_tokens` | Claude Token Count | API route | +| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) | +| `GET /api/models/catalog` | Catalog | All models grouped by provider + type | +| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route | +| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration | +| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint | +| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models | -旁路处理程序 (`open-sse/utils/bypassHandler.ts`) 拦截来自 Claude CLI 的已知“一次性”请求(预热 ping、标题提取和令牌计数),并返回**虚假响应**,而不消耗上游提供商令牌。仅当“User-Agent”包含“claude-cli”时才会触发。## Request Logger Pipeline +## Bypass Handler -请求记录器 (`open-sse/utils/requestLogger.ts`) 提供了一个 7 阶段调试日志记录管道,默认情况下禁用,通过 `ENABLE_REQUEST_LOGS=true` 启用:``` +The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`. + +## Request Logger Pipeline + +The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`: + +``` 1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json → 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt -```` +``` -每个请求会话的文件都会写入“/logs//”。## Failure Modes and Resilience +Files are written to `/logs//` for each request session. + +## Failure Modes and Resilience ## 1) Account/Provider Availability -- 提供商帐户因瞬态/速率/身份验证错误而冷却 -- 请求失败之前的帐户回退 -- 当前模型/提供商路径耗尽时组合模型回退## 2) Token Expiry +- provider account cooldown on transient/rate/auth errors +- account fallback before failing request +- combo model fallback when current model/provider path is exhausted -- 对可刷新提供程序进行预检查和刷新并重试 -- 401/403 在核心路径中尝试刷新后重试## 3) Stream Safety +## 2) Token Expiry -- 断开连接感知流控制器 -- 带有流尾刷新和“[DONE]”处理的翻译流 -- 当提供者使用元数据丢失时使用估计回退## 4) Cloud Sync Degradation +- pre-check and refresh with retry for refreshable providers +- 401/403 retry after refresh attempt in core path -- 出现同步错误,但本地运行时仍在继续 -- 调度程序具有可重试的逻辑,但定期执行当前默认调用单次尝试同步## 5) Data Integrity +## 3) Stream Safety -- SQLite 模式迁移和启动时自动升级挂钩 -- 遗留 JSON → SQLite 迁移兼容性路径## Observability and Operational Signals +- disconnect-aware stream controller +- translation stream with end-of-stream flush and `[DONE]` handling +- usage estimation fallback when provider usage metadata is missing -运行时可见性来源: +## 4) Cloud Sync Degradation -- 来自`src/sse/utils/logger.ts`的控制台日志 -- SQLite 中每个请求的使用情况聚合(`usage_history`、`call_logs`、`proxy_logs`) -- 当“settings.detailed_logs_enabled=true”时,SQLite 中的四阶段详细有效负载捕获(“request_detail_logs”) -- “log.txt”中的文本请求状态日志(可选/兼容) -- 当“ENABLE_REQUEST_LOGS=true”时,可选的深度请求/翻译日志位于“logs/”下 -- 用于 UI 使用的仪表板使用端点 (`/api/usage/*`) +- sync errors are surfaced but local runtime continues +- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default -详细的请求有效负载捕获为每个路由调用存储最多四个 JSON 有效负载阶段: +## 5) Data Integrity -- 从客户端收到的原始请求 -- 翻译后的请求实际发送到上游 -- 提供商响应重构为 JSON;流式响应被压缩为最终摘要加上流元数据 -- OmniRoute 返回的最终客户端响应;流式响应以相同的紧凑摘要形式存储## Security-Sensitive Boundaries +- SQLite schema migrations and auto-upgrade hooks at startup +- legacy JSON → SQLite migration compatibility path -- JWT 秘密 (`JWT_SECRET`) 确保仪表板会话 cookie 验证/签名 -- 应为首次运行配置显式配置初始密码引导程序(“INITIAL_PASSWORD”) -- API 密钥 HMAC 秘密 (`API_KEY_SECRET`) 确保生成的本地 API 密钥格式 -- 提供者机密(API 密钥/令牌)保留在本地数据库中,并应在文件系统级别受到保护 -- 云同步端点依赖于 API 密钥身份验证 + 机器 ID 语义## Environment and Runtime Matrix +## Observability and Operational Signals -代码主动使用的环境变量: +Runtime visibility sources: -- 应用程序/身份验证:`JWT_SECRET`、`INITIAL_PASSWORD` -- 存储:`DATA_DIR` -- 兼容的节点行为:`ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` -- 可选的存储基础覆盖(Linux/macOS 当 `DATA_DIR` 未设置时):`XDG_CONFIG_HOME` -- 安全哈希:`API_KEY_SECRET`、`MACHINE_ID_SALT` -- 日志记录:`ENABLE_REQUEST_LOGS` -- 同步/云 URL:`NEXT_PUBLIC_BASE_URL`、`NEXT_PUBLIC_CLOUD_URL` -- 出站代理:`HTTP_PROXY`、`HTTPS_PROXY`、`ALL_PROXY`、`NO_PROXY` 和小写变体 -- SOCKS5 功能标志:`ENABLE_SOCKS5_PROXY`、`NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` -- 平台/运行时帮助程序(不是特定于应用程序的配置):`APPDATA`、`NODE_ENV`、`PORT`、`HOSTNAME`## Known Architectural Notes +- console logs from `src/sse/utils/logger.ts` +- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`) +- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true` +- textual request status log in `log.txt` (optional/compat) +- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true` +- dashboard usage endpoints (`/api/usage/*`) for UI consumption -1. `usageDb` 和 `localDb` 与旧文件迁移共享相同的基本目录策略 (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`)。 -2. `/api/v1/route.ts` 委托给 `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) 使用的同一统一目录构建器,以避免语义漂移。 -3. 请求记录器在启用时写入完整的标头/正文;将日志目录视为敏感目录。 -4. 云行为取决于正确的“NEXT_PUBLIC_BASE_URL”和云端点可访问性。 -5. `open-sse/` 目录发布为 `@omniroute/open-sse`**npm 工作区包**。源代码通过 `@omniroute/open-sse/...` 导入它(由 Next.js `transpilePackages` 解析)。为了保持一致性,本文档中的文件路径仍然使用目录名称“open-sse/”。 -6. 仪表板中的图表使用**Recharts**(基于 SVG)来实现可访问的交互式分析可视化(模型使用情况条形图、包含成功率的提供商细分表)。 -7. E2E 测试使用**Playwright**(`tests/e2e/`),通过 `npm run test:e2e` 运行。单元测试使用**Node.js 测试运行程序**(`tests/unit/`),通过 `npm run test:unit` 运​​行。 `src/` 下的源代码是**TypeScript**(`.ts`/`.tsx`); `open-sse/` 工作区仍然是 JavaScript (`.js`)。 -8. 设置页面分为 5 个选项卡:安全、路由(6 种全局策略:先填充、循环、p2c、随机、最少使用、成本优化)、弹性(可编辑速率限制、断路器、策略)、AI(思考预算、系统提示、提示缓存)、高级(代理)。## Operational Verification Checklist +Detailed request payload capture stores up to four JSON payload stages per routed call: -- 从源代码构建:`npm run build` -- 构建 Docker 镜像:“docker build -tomniroute”。 -- 启动服务并验证: -- `获取/api/设置` +- raw request received from the client +- translated request actually sent upstream +- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata +- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form + +## Security-Sensitive Boundaries + +- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing +- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning +- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format +- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level +- Cloud sync endpoints rely on API key auth + machine id semantics + +## Environment and Runtime Matrix + +Environment variables actively used by code: + +- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD` +- Storage: `DATA_DIR` +- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE` +- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME` +- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT` +- Logging: `ENABLE_REQUEST_LOGS` +- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL` +- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants +- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY` +- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME` + +## Known Architectural Notes + +1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration. +2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift. +3. Request logger writes full headers/body when enabled; treat log directory as sensitive. +4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability. +5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency. +6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates). +7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`). +8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy). +9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed. +10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22. +11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible. + +## Operational Verification Checklist + +- Build from source: `npm run build` +- Build Docker image: `docker build -t omniroute .` +- Start service and verify: +- `GET /api/settings` - `GET /api/v1/models` -- 当“PORT=20128”时,CLI 目标基本 URL 应为“http://:20128/v1” +- CLI target base URL should be `http://:20128/v1` when `PORT=20128` diff --git a/docs/i18n/zh-CN/docs/FEATURES.md b/docs/i18n/zh-CN/docs/FEATURES.md index da02cc9693..2f633f8859 100644 --- a/docs/i18n/zh-CN/docs/FEATURES.md +++ b/docs/i18n/zh-CN/docs/FEATURES.md @@ -4,102 +4,168 @@ --- -OmniRoute 仪表板每个部分的视觉指南。--- + + +Visual guide to every section of the OmniRoute dashboard. + +--- ## 🔌 Providers -管理 AI 提供商连接:OAuth 提供商(Claude Code、Codex、Gemini CLI)、API 密钥提供商(Groq、DeepSeek、OpenRouter)和免费提供商(Qoder、Qwen、Kiro)。 Kiro 账户包括信用余额跟踪 — 剩余信用、总限额和续订日期可在仪表板 → 使用情况中查看。![Providers Dashboard](screenshots/01-providers.png) +Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage. + +![Providers Dashboard](screenshots/01-providers.png) --- ## 🎨 Combos -使用 6 种策略创建模型路由组合:优先级、加权、循环、随机、最少使用和成本优化。每个组合都通过自动回退链接多个模型,并包括快速模板和准备情况检查。![Combos Dashboard](screenshots/02-combos.png) +Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks. + +![Combos Dashboard](screenshots/02-combos.png) --- ## 📊 Analytics -全面的使用分析,包括代币消耗、成本估算、活动热图、每周分布图和每个提供商的细分。![Analytics Dashboard](screenshots/03-analytics.png) +Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns. + +![Analytics Dashboard](screenshots/03-analytics.png) --- ## 🏥 System Health -实时监控:正常运行时间、内存、版本、延迟百分位数 (p50/p95/p99)、缓存统计数据和提供商断路器状态。![Health Dashboard](screenshots/04-health.png) +Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states. + +![Health Dashboard](screenshots/04-health.png) --- ## 🔧 Translator Playground -用于调试 API 翻译的四种模式:**Playground**(格式转换器)、**Chat Tester**(实时请求)、**Test Bench**(批量测试)和**Live Monitor**(实时流)。![Translator Playground](screenshots/05-translator.png) +Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream). + +![Translator Playground](screenshots/05-translator.png) --- ## 🎮 Model Playground _(v2.0.9+)_ -Test any model directly from the dashboard.选择提供商、模型和端点,使用 Monaco 编辑器编写提示、实时流式传输响应、中止中流以及查看计时指标。--- +Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics. + +--- ## 🎨 Themes _(v2.0.5+)_ -整个仪表板的可定制颜色主题。从 7 种预设颜色(珊瑚色、蓝色、红色、绿色、紫色、橙色、青色)中进行选择,或通过选择任何十六进制颜色来创建自定义主题。支持浅色、深色和系统模式。--- +Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode. + +--- ## ⚙️ Settings -带选项卡的综合设置面板: +Comprehensive settings panel with tabs: --**常规**— 系统存储、备份管理(导出/导入数据库)-**外观**- 主题选择器(深色/浅色/系统)、颜色主题预设和自定义颜色、运行状况日志可见性、侧边栏项目可见性控制 -**安全**— API 端点保护、自定义提供商阻止、IP 过滤、会话信息 -**路由**— 模型别名、后台任务降级 -**弹性**- 速率限制持久性、断路器调整、自动禁用被禁止的帐户、提供商到期监控 -**高级**— 配置覆盖、配置审计跟踪、回退降级模式![Settings Dashboard](screenshots/06-settings.png) +- **General** — System storage, backup management (export/import database) +- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls +- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info +- **Routing** — Model aliases, background task degradation +- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration +- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode + +![Settings Dashboard](screenshots/06-settings.png) --- ## 🔧 CLI Tools -一键配置 AI 编码工具:Claude Code、Codex CLI、Gemini CLI、OpenClaw、Kilo Code、Antigravity、Cline、Continue、Cursor 和 Factory Droid。具有自动配置应用/重置、连接配置文件和模型映射功能。![CLI Tools Dashboard](screenshots/07-cli-tools.png) +One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping. + +![CLI Tools Dashboard](screenshots/07-cli-tools.png) --- ## 🤖 CLI Agents _(v2.0.11+)_ -用于发现和管理 CLI 代理的仪表板。显示 14 个内置代理(Codex、Claude、Goose、Gemini CLI、OpenClaw、Aider、OpenCode、Cline、Qwen Code、ForgeCode、Amazon Q、Open Interpreter、Cursor CLI、Warp)的网格,其中: +Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with: --**安装状态**— 已安装/未通过版本检测找到 -**协议徽章**— stdio、HTTP 等。-**自定义代理**— 通过表单注册任何 CLI 工具(名称、二进制文件、版本命令、spawn args)-**CLI 指纹匹配**— 每个提供商切换以匹配本机 CLI 请求签名,在保留代理 IP 的同时降低禁令风险--- +- **Installation status** — Installed / Not Found with version detection +- **Protocol badges** — stdio, HTTP, etc. +- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args) +- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP + +--- + +## 🔗 Context Relay _(v3.5.5+)_ + +A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context. + +Configurable via combo-level or global settings: +- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%) +- **Max Messages For Summary** — How much recent history to condense +- **Summary Model** — Optional override model for generating the handoff summary + +Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md). + +--- + +## 🛡️ Proxy Hardening _(v3.5.5+)_ + +Comprehensive proxy configuration enforcement across the entire request pipeline: + +- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments +- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings +- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22 +- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS + +--- ## 🖼️ Media _(v2.0.3+)_ -从仪表板生成图像、视频和音乐。支持 OpenAI、xAI、Together、Hyperbolic、SD WebUI、ComfyUI、AnimateDiff、Stable Audio Open 和 MusicGen。--- +Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen. + +--- ## 📝 Request Logs -实时请求记录,并按提供商、模型、帐户和 API 密钥进行过滤。显示状态代码、令牌使用情况、延迟和响应详细信息。![Usage Logs](screenshots/08-usage.png) +Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details. + +![Usage Logs](screenshots/08-usage.png) --- ## 🌐 API Endpoint -您的统一 API 端点具有功能细分:聊天完成、响应 API、嵌入、图像生成、重新排名、音频转录、文本转语音、审核和注册 API 密钥。 Cloudflare Quick Tunnel 集成和云代理支持远程访问。![Endpoint Dashboard](screenshots/09-endpoint.png) +Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access. + +![Endpoint Dashboard](screenshots/09-endpoint.png) --- ## 🔑 API Key Management -创建、范围和撤销 API 密钥。每个密钥都可以限制为具有完全访问或只读权限的特定模型/提供商。具有使用跟踪功能的可视化密钥管理。--- +Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking. + +--- ## 📋 Audit Log -管理操作跟踪,可按操作类型、参与者、目标、IP 地址和时间戳进行过滤。完整的安全事件历史记录。--- +Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history. + +--- ## 🖥️ Desktop Application -适用于 Windows、macOS 和 Linux 的本机 Electron 桌面应用程序。将 OmniRoute 作为独立应用程序运行,具有系统托盘集成、离线支持、自动更新和一键安装功能。 +Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install. -主要特点: +Key features: -- 服务器就绪轮询(冷启动时无空白屏幕) -- 带端口管理的系统托盘 -- 内容安全政策 -- 单实例锁 -- 重启时自动更新 -- 平台条件 UI(macOS 红绿灯、Windows/Linux 默认标题栏) -- 强化 Electron 构建打包 - 在打包之前检测并拒绝独立包中的符号链接“node_modules”,从而防止运行时对构建机器的依赖(v2.5.5+) +- Server readiness polling (no blank screen on cold start) +- System tray with port management +- Content Security Policy +- Single-instance lock +- Auto-update on restart +- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar) +- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+) -📖 请参阅 [`electron/README.md`](../electron/README.md) 以获取完整文档。 +📖 See [`electron/README.md`](../electron/README.md) for full documentation. diff --git a/docs/i18n/zh-CN/docs/TROUBLESHOOTING.md b/docs/i18n/zh-CN/docs/TROUBLESHOOTING.md index 267f2ce7c8..d9aaae4b22 100644 --- a/docs/i18n/zh-CN/docs/TROUBLESHOOTING.md +++ b/docs/i18n/zh-CN/docs/TROUBLESHOOTING.md @@ -4,65 +4,142 @@ --- -OmniRoute 的常见问题和解决方案。--- + + +Common problems and solutions for OmniRoute. + +--- ## Quick Fixes -| 问题 | 解决方案 | -| ---------------------- | ------------------------------------------------------------------ | --- | -| 首次登录无法使用 | 在`.env`中设置`INITIAL_PASSWORD`(无硬编码默认值) | -| 仪表板在错误端口上打开 | 设置 `PORT=20128` 和 `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | -| `logs/` 下没有请求日志 | 设置`ENABLE_REQUEST_LOGS=true` | -| EACCES:权限被拒绝 | 设置 `DATA_DIR=/path/to/writable/dir` 以覆盖 `~/.omniroute` | -| 路由策略未保存 | 更新至 v1.4.11+(Zod 架构修复设置持久性) | --- | +| Problem | Solution | +| ----------------------------- | ------------------------------------------------------------------ | +| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) | +| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` | +| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` | +| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` | +| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) | +| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below | +| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below | + +--- + +## Node.js Compatibility + + + +### Login page crashes or shows "Module self-registration" error + +**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database. + +**Symptoms:** +- Login page shows a blank screen or a server error +- Console shows `Error: Module did not self-register` or similar native binding errors +- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected + +**Fix:** + +1. Install Node.js 22 LTS (recommended): + ```bash + nvm install 22 + nvm use 22 + ``` +2. Verify your version: `node --version` should show `v22.x.x` +3. Reinstall OmniRoute: `npm install -g omniroute` +4. Restart: `omniroute` + +> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**. + +--- + +## Proxy Issues + + + +### Provider validation shows "fetch failed" + +**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing. + +**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically. + +### Token health check fails with "fetch failed" + +**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection. + +**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+. + +### SOCKS5 proxy returns "invalid onRequestStart method" + +**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation. + +**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+. + +--- ## Provider Issues ### "Language model did not provide messages" -**原因:**提供商配额已用完。 +**Cause:** Provider quota exhausted. -**修复:** +**Fix:** -1.检查仪表板配额跟踪器2. 使用具有后备层的组合3.切换到更便宜/免费的套餐### Rate Limiting +1. Check dashboard quota tracker +2. Use a combo with fallback tiers +3. Switch to cheaper/free tier -**原因:**订阅配额已用完。 +### Rate Limiting -**修复:** +**Cause:** Subscription quota exhausted. -- 添加后备:`cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` -- 使用 GLM/MiniMax 作为廉价备份### OAuth Token Expired +**Fix:** -OmniRoute 自动刷新令牌。如果问题仍然存在: +- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking` +- Use GLM/MiniMax as cheap backup -1. 仪表板 → 提供商 → 重新连接 2.删除并重新添加提供商连接--- +### OAuth Token Expired + +OmniRoute auto-refreshes tokens. If issues persist: + +1. Dashboard → Provider → Reconnect +2. Delete and re-add the provider connection + +--- ## Cloud Issues ### Cloud Sync Errors -1. 验证“BASE_URL”指向您正在运行的实例(例如“http://localhost:20128”) -2. 验证“CLOUD_URL”指向您的云端点(例如“https://omniroute.dev”) -3. 保持“NEXT*PUBLIC*\*”值与服务器端值一致### Cloud `stream=false` Returns 500 +1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`) +2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`) +3. Keep `NEXT_PUBLIC_*` values aligned with server-side values -**症状:**“非流式调用的云端点上出现意外的令牌“d”...”。 +### Cloud `stream=false` Returns 500 -**原因:**上游返回 SSE 负载,而客户端需要 JSON。 +**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls. -**解决方法:**使用 `stream=true` 进行云直接调用。本地运行时包括 SSE→JSON 回退。### Cloud Says Connected but "Invalid API key" +**Cause:** Upstream returns SSE payload while client expects JSON. -1. 从本地仪表板创建新密钥 (`/api/keys`) -2. 运行云同步:启用云→立即同步 -3. 旧的/未同步的密钥仍然可以在云上返回“401”--- +**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback. + +### Cloud Says Connected but "Invalid API key" + +1. Create a fresh key from local dashboard (`/api/keys`) +2. Run cloud sync: Enable Cloud → Sync Now +3. Old/non-synced keys can still return `401` on cloud + +--- ## Docker Issues ### CLI Tool Shows Not Installed -1. 检查运行时字段: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` -2. 对于便携模式:使用镜像目标“runner-cli”(捆绑的 CLI) -3. 对于主机挂载模式:设置 `CLI_EXTRA_PATHS` 并将主机 bin 目录挂载为只读 -4. 如果“installed=true”且“runnable=false”:已找到二进制文件,但运行状况检查失败### Quick Runtime Validation +1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq` +2. For portable mode: use image target `runner-cli` (bundled CLIs) +3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only +4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck + +### Quick Runtime Validation ```bash curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}' @@ -76,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed, ### High Costs -1. 在 Dashboard → 使用情况中查看使用情况统计数据 -2. 将主模型切换为GLM/MiniMax +1. Check usage stats in Dashboard → Usage +2. Switch primary model to GLM/MiniMax 3. Use free tier (Gemini CLI, Qoder) for non-critical tasks -4. Set cost budgets per API key: Dashboard → API Keys → Budget--- +4. Set cost budgets per API key: Dashboard → API Keys → Budget + +--- ## Debugging ### Enable Request Logs -在“.env”文件中设置“ENABLE_REQUEST_LOGS=true”。日志显示在“logs/”目录下。### Check Provider Health +Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory. + +### Check Provider Health ```bash # Health dashboard @@ -97,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health ### Runtime Storage -- 主状态:`${DATA_DIR}/storage.sqlite`(提供程序、组合、别名、键、设置) -- 用法:`storage.sqlite` 中的 SQLite 表(`usage_history`、`call_logs`、`proxy_logs`)+ 可选的 `${DATA_DIR}/log.txt` 和 `${DATA_DIR}/call_logs/` -- 请求日志:`/logs/...`(当`ENABLE_REQUEST_LOGS=true`时)--- +- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings) +- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/` +- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`) + +--- ## Circuit Breaker Issues ### Provider stuck in OPEN state -当提供商的断路器打开时,请求将被阻止,直到冷却时间到期。 +When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires. -**修复:** +**Fix:** -1. 转到**仪表板 → 设置 → 弹性** -2. 检查受影响提供商的断路器卡 -3. 单击“**全部重置**”以清除所有断路器,或等待冷却时间到期 -4. 重置前验证提供商是否确实可用### Provider keeps tripping the circuit breaker +1. Go to **Dashboard → Settings → Resilience** +2. Check the circuit breaker card for the affected provider +3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire +4. Verify the provider is actually available before resetting -如果提供者重复进入 OPEN 状态: +### Provider keeps tripping the circuit breaker -1. 检查**仪表板 → 运行状况 → 提供商运行状况**以了解故障模式 -2. 转到**设置 → 恢复能力 → 提供商配置文件**并增加失败阈值 -3. 检查提供商是否更改了 API 限制或需要重新身份验证 -4. 检查延迟遥测 — 高延迟可能会导致基于超时的故障--- +If a provider repeatedly enters OPEN state: + +1. Check **Dashboard → Health → Provider Health** for the failure pattern +2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold +3. Check if the provider has changed API limits or requires re-authentication +4. Review latency telemetry — high latency may cause timeout-based failures + +--- ## Audio Transcription Issues ### "Unsupported model" error -- 确保您使用正确的前缀:“deepgram/nova-3”或“assembleai/best” -- 验证提供商是否已在**仪表板 → 提供商**中连接### Transcription returns empty or fails +- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best` +- Verify the provider is connected in **Dashboard → Providers** -- 检查支持的音频格式:`mp3`、`wav`、`m4a`、`flac`、`ogg`、`webm` -- 验证文件大小是否在提供商限制内(通常< 25MB) -- 检查提供商卡中提供商 API 密钥的有效性--- +### Transcription returns empty or fails + +- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm` +- Verify file size is within provider limits (typically < 25MB) +- Check provider API key validity in the provider card + +--- ## Translator Debugging -使用**Dashboard → Translator**调试格式转换问题: +Use **Dashboard → Translator** to debug format translation issues: -| 模式 | 何时使用 | -| -------------- | ------------------------------------------------------ | ------------------------ | -| **游乐场** | 并排比较输入/输出格式 — 粘贴失败的请求以查看其如何翻译 | -| **聊天测试仪** | 发送实时消息并检查完整的请求/响应负载(包括标头) | -| **测试台** | 跨格式组合运行批量测试以查找哪些翻译被破坏 | -| **实时监控** | 观看实时请求流以捕获间歇性翻译问题 | ### Common format issues | +| Mode | When to Use | +| ---------------- | -------------------------------------------------------------------------------------------- | +| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates | +| **Chat Tester** | Send live messages and inspect the full request/response payload including headers | +| **Test Bench** | Run batch tests across format combinations to find which translations are broken | +| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues | --**思维标签未出现**— 检查目标提供商是否支持思维以及思维预算设置 -**工具调用丢失**— 某些格式翻译可能会删除不支持的字段;在 Playground 模式下验证 -**系统提示缺失**— Claude 和 Gemini 处理系统提示的方式不同;检查翻译输出 -**SDK 返回原始字符串而不是对象**— 在 v1.1.0 中修复:响应清理程序现在会删除导致 OpenAI SDK Pydantic 验证失败的非标准字段(`x_groq`、`usage_breakdown` 等)-**GLM/ERNIE 拒绝 `system` 角色**— 在 v1.1.0 中修复:角色规范器自动将系统消息合并到不兼容模型的用户消息中 -**`开发人员`角色无法识别**— v1.1.0 中修复:对于非 OpenAI 提供商自动转换为`系统` -**`json_schema` 不适用于 Gemini**— 在 v1.1.0 中修复:`response_format` 现在转换为 Gemini 的 `responseMimeType` + `responseSchema`--- +### Common format issues + +- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting +- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode +- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output +- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures +- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models +- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers +- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema` + +--- ## Resilience Settings ### Auto rate-limit not triggering -- 自动速率限制仅适用于 API 密钥提供商(不适用于 OAuth/订阅) -- 验证**设置 → 弹性 → 提供商配置文件**已启用自动速率限制 -- 检查提供商是否返回“429”状态代码或“Retry-After”标头### Tuning exponential backoff +- Auto rate-limit only applies to API key providers (not OAuth/subscription) +- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled +- Check if the provider returns `429` status codes or `Retry-After` headers -提供商配置文件支持以下设置: +### Tuning exponential backoff --**基本延迟**— 第一次失败后的初始等待时间(默认值:1 秒)-**最大延迟**— 最大等待时间上限(默认值:30 秒)-**乘数**— 每次连续失败增加多少延迟(默认值:2x)### Anti-thundering herd +Provider profiles support these settings: -当许多并发请求到达速率受限的提供程序时,OmniRoute 使用互斥锁 + 自动速率限制来序列化请求并防止级联故障。对于 API 密钥提供者来说,这是自动的。--- +- **Base delay** — Initial wait time after first failure (default: 1s) +- **Max delay** — Maximum wait time cap (default: 30s) +- **Multiplier** — How much to increase delay per consecutive failure (default: 2x) + +### Anti-thundering herd + +When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers. + +--- ## Optional RAG / LLM failure taxonomy (16 problems) -一些 OmniRoute 用户将网关放置在 RAG 或代理堆栈前面。在这些设置中,经常会看到一种奇怪的模式:OmniRoute 看起来很健康(提供程序正常,路由配置文件正常,没有速率限制警报),但最终答案仍然是错误的。 +Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong. -实际上,这些事件通常来自下游 RAG 管道,而不是来自网关本身。 +In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself. -如果您想要一个共享词汇表来描述这些故障,您可以使用 WFGY ProblemMap,这是一个外部 MIT 许可证文本资源,定义了 16 种重复出现的 RAG / LLM 故障模式。从高层次来看,它涵盖: +If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers: -- 检索漂移和打破上下文边界 -- 空或过时的索引和向量存储 -- 嵌入与语义不匹配 -- 提示汇编和上下文窗口问题 -- 逻辑崩溃和过于自信的答案 -- 长链和代理协调失败 -- 多代理记忆和角色漂移 -- 部署和引导排序问题 +- retrieval drift and broken context boundaries +- empty or stale indexes and vector stores +- embedding versus semantic mismatch +- prompt assembly and context window issues +- logic collapse and overconfident answers +- long chain and agent coordination failures +- multi agent memory and role drift +- deployment and bootstrap ordering problems -这个想法很简单: +The idea is simple: -1. 当您调查不良响应时,捕获: - - 用户任务和请求 - - OmniRoute 中的路线或提供商组合 - - 下游使用的任何 RAG 上下文(检索的文档、工具调用等) -2. 将事件映射到一两个 WFGY ProblemMap 编号(“No.1”…“No.16”)。 -3. 将号码存储在您自己的仪表板、运行手册或 OmniRoute 日志旁边的事件跟踪器中。 -4. 使用相应的 WFGY 页面来决定是否需要更改 RAG 堆栈、检索器或路由策略。 +1. When you investigate a bad response, capture: + - user task and request + - route or provider combo in OmniRoute + - any RAG context used downstream (retrieved documents, tool calls, etc) +2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`). +3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs. +4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy. -全文和具体食谱在这里(麻省理工学院许可证,仅限文本): +Full text and concrete recipes live here (MIT license, text only): -[WFGY 问题地图自述文件](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) +[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md) -如果您不在 OmniRoute 后面运行 RAG 或代理管道,则可以忽略此部分。--- +You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute. + +--- ## Still Stuck? --**GitHub 问题**:[github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**架构**:请参阅 [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) 了解内部详细信息 -**API 参考**:请参阅 [`docs/API_REFERENCE.md`](API_REFERENCE.md) 了解所有端点 -**健康仪表板**:检查**仪表板→健康**以获取实时系统状态 -**翻译器**:使用**仪表板→翻译器**来调试格式问题 +- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) +- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details +- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints +- **Health Dashboard**: Check **Dashboard → Health** for real-time system status +- **Translator**: Use **Dashboard → Translator** to debug format issues diff --git a/docs/i18n/zh-CN/llm.txt b/docs/i18n/zh-CN/llm.txt new file mode 100644 index 0000000000..6b2c51cc77 --- /dev/null +++ b/docs/i18n/zh-CN/llm.txt @@ -0,0 +1,476 @@ +# OmniRoute (中文(简体)) + +🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt) + +--- + + +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app. + +## 概述 + +OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free. + +**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. + +**Current version:** 3.5.5 + +## Tech Stack + +- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`) +- **Framework:** Next.js 16 (App Router) with TypeScript 5.9 +- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations) +- **State management:** Zustand (client), SQLite (server persistence) +- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons +- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth +- **Schemas:** Zod v4 for all API / MCP input validation +- **Background jobs:** Custom token health check scheduler, 24h model auto-sync +- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses +- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine +- **i18n:** next-intl with 30 languages +- **Desktop:** Electron (cross-platform: Windows, macOS, Linux) +- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`) + +## Project Structure + +``` +/ +├── src/ # Main application source +│ ├── app/ # Next.js App Router pages and API routes +│ │ ├── (dashboard)/ # Dashboard UI pages +│ │ │ └── dashboard/ +│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents) +│ │ │ ├── analytics/ # Usage analytics and charts +│ │ │ ├── api-manager/ # API key management +│ │ │ ├── audit/ # Audit logs +│ │ │ ├── auto-combo/ # Auto-combo engine dashboard +│ │ │ ├── cache/ # Cache dashboard (semantic cache stats) +│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) +│ │ │ ├── costs/ # Cost tracking per provider/model +│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs +│ │ │ ├── health/ # System health (uptime, circuit breakers, latency) +│ │ │ ├── limits/ # Rate limits dashboard +│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed) +│ │ │ ├── media/ # Image/video/music generation + transcription +│ │ │ ├── memory/ # Memory system dashboard +│ │ │ ├── onboarding/ # Onboarding wizard +│ │ │ ├── playground/ # Model playground (Monaco editor, streaming) +│ │ │ ├── providers/ # Provider management (OAuth + API key + free) +│ │ │ ├── search-tools/ # Search tools configuration +│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced) +│ │ │ ├── skills/ # Skills system dashboard +│ │ │ ├── translator/ # Format translator + debug tools +│ │ │ └── usage/ # Usage history +│ │ ├── api/ # REST API endpoints (51 route directories) +│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings, +│ │ │ │ # images, audio, videos, music, moderations, rerank, search, +│ │ │ │ # responses, messages, registered-keys, quotas, accounts) +│ │ │ ├── v1beta/ # Gemini-compatible API +│ │ │ ├── a2a/ # A2A agent management API +│ │ │ ├── acp/ # ACP agent management API +│ │ │ ├── oauth/ # OAuth flows per provider +│ │ │ ├── providers/ # Provider CRUD and batch testing +│ │ │ ├── models/ # Dashboard model listing and aliases +│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains) +│ │ │ ├── memory/ # Memory system API +│ │ │ ├── skills/ # Skills system API +│ │ │ ├── evals/ # Eval runner API +│ │ │ ├── mcp/ # MCP HTTP transport API +│ │ │ ├── search/ # Search provider API +│ │ │ ├── webhooks/ # Webhook management +│ │ │ ├── tunnels/ # Cloudflare tunnel management +│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.) +│ │ ├── landing/ # Landing page +│ │ ├── login/ # Login page +│ │ ├── forgot-password/ # Password recovery +│ │ ├── status/ # Status page +│ │ └── docs/ # In-app documentation +│ ├── domain/ # Domain types and policy engine +│ │ ├── policyEngine.ts # Central policy engine +│ │ ├── comboResolver.ts # Combo resolution logic +│ │ ├── costRules.ts # Cost calculation rules +│ │ ├── degradation.ts # Graceful degradation +│ │ ├── fallbackPolicy.ts # Fallback behavior +│ │ ├── lockoutPolicy.ts # Account lockout logic +│ │ ├── modelAvailability.ts # Model availability checks +│ │ ├── providerExpiration.ts # Provider credential expiration +│ │ ├── quotaCache.ts # Quota caching layer +│ │ ├── configAudit.ts # Configuration auditing +│ │ └── responses.ts # Domain response types +│ ├── i18n/ # Internationalization +│ │ └── messages/ # 30 language JSON files +│ ├── lib/ # Core libraries +│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server +│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting) +│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup +│ │ │ └── streaming.ts # SSE streaming for A2A +│ │ ├── acp/ # Agent Communication Protocol registry and manager +│ │ ├── compliance/ # Compliance policy engine +│ │ ├── db/ # SQLite database layer (21 modules + migrations) +│ │ │ ├── core.ts # Database initialization, connection, schema +│ │ │ ├── providers.ts # Provider connection CRUD +│ │ │ ├── models.ts # Model catalog management +│ │ │ ├── combos.ts # Combo configuration +│ │ │ ├── apiKeys.ts # API key management +│ │ │ ├── settings.ts # Settings persistence +│ │ │ ├── backup.ts # Database backup/restore +│ │ │ ├── proxies.ts # Proxy registry +│ │ │ ├── prompts.ts # Prompt templates +│ │ │ ├── webhooks.ts # Webhook subscriptions +│ │ │ ├── detailedLogs.ts # Detailed request logging +│ │ │ ├── domainState.ts # Domain state persistence +│ │ │ ├── registeredKeys.ts # Registered API keys with quotas +│ │ │ ├── quotaSnapshots.ts # Quota snapshot history +│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings +│ │ │ ├── cliToolState.ts # CLI tool state tracking +│ │ │ ├── encryption.ts # Data encryption +│ │ │ ├── readCache.ts # Read-through cache layer +│ │ │ ├── secrets.ts # Secrets management +│ │ │ ├── stateReset.ts # State reset utilities +│ │ │ ├── migrationRunner.ts # Schema migration runner +│ │ │ └── migrations/ # 16 SQL migration files +│ │ ├── evals/ # Eval runner and scheduler +│ │ ├── memory/ # Persistent conversational memory +│ │ │ ├── extraction.ts # Memory extraction from conversations +│ │ │ ├── injection.ts # Memory injection into context +│ │ │ ├── retrieval.ts # Memory retrieval/search +│ │ │ ├── store.ts # Memory persistence layer +│ │ │ └── summarization.ts # Memory summarization +│ │ ├── oauth/ # OAuth providers, services, and utilities +│ │ │ ├── constants/ # Default OAuth credentials (overridable via env) +│ │ │ ├── providers/ # Provider-specific OAuth configs +│ │ │ ├── services/ # Provider-specific token exchange logic +│ │ │ └── utils/ # PKCE, callback server, token helpers +│ │ ├── plugins/ # Plugin system +│ │ ├── skills/ # Extensible skill framework +│ │ │ ├── registry.ts # Skill registration +│ │ │ ├── executor.ts # Skill execution engine +│ │ │ ├── sandbox.ts # Skill sandbox environment +│ │ │ ├── builtin/ # Built-in skills +│ │ │ ├── interception.ts # Skill request interception +│ │ │ └── injection.ts # Skill context injection +│ │ ├── usage/ # Usage tracking system +│ │ │ ├── callLogs.ts # Call log persistence +│ │ │ ├── costCalculator.ts # Cost calculation engine +│ │ │ └── usageHistory.ts # Usage history queries +│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers +│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management +│ │ ├── pricingSync.ts # LiteLLM pricing data sync +│ │ ├── semanticCache.ts # Semantic caching layer +│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler +│ │ ├── webhookDispatcher.ts # Webhook event dispatcher +│ │ └── localDb.ts # Unified re-export layer for all DB modules +│ ├── middleware/ # Request middleware +│ │ └── promptInjectionGuard.ts # Prompt injection detection +│ ├── mitm/ # MITM proxy capability +│ │ ├── cert/ # Certificate management +│ │ ├── dns/ # DNS handling +│ │ ├── targets/ # Target routing +│ │ └── manager.ts # MITM proxy manager +│ ├── shared/ # Shared utilities, components, and constants +│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.) +│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes +│ │ ├── contracts/ # Shared API contracts +│ │ ├── hooks/ # React hooks +│ │ ├── middleware/ # Shared middleware utilities +│ │ ├── schemas/ # Shared Zod schemas +│ │ ├── services/ # Shared services +│ │ ├── types/ # Shared TypeScript types +│ │ ├── validation/ # Zod schemas (settings, providers, routes) +│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID) +│ ├── sse/ # SSE proxy pipeline +│ │ ├── services/ # Auth resolution, format translation, response handling +│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency +│ ├── store/ # Zustand client-side stores (theme, providers, etc.) +│ └── types/ # TypeScript type definitions +├── open-sse/ # Standalone SSE server (npm workspace) +│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video, +│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models) +│ ├── executors/ # Provider-specific request executors (14 executors) +│ │ ├── base.ts # Base executor with shared logic +│ │ ├── default.ts # Default OpenAI-compatible executor +│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum) +│ │ ├── codex.ts # OpenAI Codex CLI +│ │ ├── antigravity.ts # Antigravity IDE +│ │ ├── github.ts # GitHub Copilot +│ │ ├── gemini-cli.ts # Gemini CLI +│ │ ├── kiro.ts # Kiro AI +│ │ ├── qoder.ts # Qoder AI +│ │ ├── vertex.ts # Vertex AI (Service Account JSON) +│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI +│ │ ├── opencode.ts # OpenCode Zen/Go +│ │ ├── pollinations.ts # Pollinations AI +│ │ └── puter.ts # Puter AI +│ ├── handlers/ # Request handlers per API type (11 handlers) +│ │ ├── chatCore.ts # Main chat completions handler +│ │ ├── responsesHandler.ts # OpenAI Responses API handler +│ │ ├── embeddings.ts # Embedding generation +│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.) +│ │ ├── videoGeneration.ts # Video generation +│ │ ├── musicGeneration.ts # Music generation +│ │ ├── audioSpeech.ts # Text-to-speech +│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI) +│ │ ├── moderations.ts # Content moderation +│ │ ├── rerank.ts # Reranking API +│ │ └── search.ts # Web search API +│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP) +│ │ ├── server.ts # MCP server core (tool registration, scope enforcement) +│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools) +│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a) +│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes) +│ │ ├── audit.ts # Tool call audit logging +│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat +│ │ └── httpTransport.ts # HTTP transport handler +│ ├── services/ # 36+ service modules +│ │ ├── combo.ts # Core routing engine +│ │ ├── usage.ts # Usage tracking +│ │ ├── tokenRefresh.ts # OAuth token refresh +│ │ ├── rateLimitManager.ts # Rate limit management +│ │ ├── accountFallback.ts # Multi-account fallback +│ │ ├── sessionManager.ts # Session management +│ │ ├── wildcardRouter.ts # Wildcard model routing +│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration) +│ │ ├── intentClassifier.ts # Request intent classification +│ │ ├── taskAwareRouter.ts # Task-aware routing +│ │ ├── thinkingBudget.ts # Thinking budget management +│ │ ├── contextManager.ts # Context window management +│ │ ├── modelDeprecation.ts # Model deprecation handling +│ │ ├── modelFamilyFallback.ts # Intra-family model fallback +│ │ ├── emergencyFallback.ts # Emergency fallback +│ │ ├── workflowFSM.ts # Workflow state machine +│ │ ├── backgroundTaskDetector.ts # Background task detection +│ │ ├── ipFilter.ts # IP-based access control +│ │ ├── signatureCache.ts # CLI signature caching +│ │ ├── volumeDetector.ts # Request volume detection +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) +│ ├── transformer/ # Responses API transformer +│ │ └── responsesTransformer.ts +│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) +│ │ ├── request/ # Request translators per provider +│ │ ├── response/ # Response translators per provider +│ │ ├── helpers/ # Translation helpers +│ │ └── image/ # Image format translation +│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.) +├── electron/ # Electron desktop app (cross-platform) +│ ├── main.js # Electron main process +│ ├── preload.js # Preload script (IPC bridge) +│ └── assets/ # App icons and assets +├── tests/ # Test suites +│ ├── unit/ # 122 unit test files +│ ├── integration/ # Integration tests +│ ├── e2e/ # Playwright E2E tests +│ ├── security/ # Security tests +│ ├── translator/ # Translator-specific tests +│ └── load/ # Load tests +├── docs/ # Documentation +│ ├── i18n/ # 30-language translated docs +│ ├── ARCHITECTURE.md # Full architecture documentation +│ ├── API_REFERENCE.md # API reference +│ ├── USER_GUIDE.md # User guide +│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview +│ ├── CLI-TOOLS.md # CLI tools integration guide +│ ├── A2A-SERVER.md # A2A agent protocol documentation +│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring) +│ ├── MCP-SERVER.md # MCP server (25 tools) +│ ├── TROUBLESHOOTING.md # Troubleshooting guide +│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide +│ ├── openapi.yaml # OpenAPI specification +│ └── screenshots/ # Dashboard screenshots +├── bin/ # CLI entry points (omniroute, reset-password) +├── scripts/ # Build and utility scripts +└── .env.example # Environment variable template +``` + +## Key Features (v3.5.5) + +### Core Proxy +- **60+ AI providers** with automatic format translation +- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay +- **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity +- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown +- **Semantic caching** with cache hit/miss headers +- **Idempotency** with configurable dedup window +- **Circuit breaker** per provider with configurable thresholds +- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback +- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers +- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement +- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization +- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills +- **Prompt Injection Guard**: Middleware-level prompt injection detection +- **MITM Proxy**: Certificate management, DNS handling, and target routing +- **Cloudflare Tunnels**: Managed tunnel creation for remote access +- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) + +### 安全 +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` +- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` +- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams +- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection +- **CLI Fingerprint Matching** — Per-provider request signature matching +- **Prompt injection guard** — Request middleware detection +- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`) +- **PII sanitizer** — Sensitive data scrubbing in logs + +### Dashboard Pages (23 sections) +- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies +- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics +- **Analytics** — Token consumption, cost, heatmaps, distributions +- **Health** — Uptime, memory, latency percentiles, circuit breakers +- **Logs** — Request, Proxy, Audit, Console (tabbed) +- **Audit** — Audit trail and compliance logging +- **Costs** — Cost tracking per provider/model +- **Limits** — Rate limit monitoring +- **Cache** — Semantic cache statistics and management +- **CLI Tools** — One-click configuration for 10+ AI CLI tools +- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration +- **Playground** — Test any model with Monaco editor, streaming responses +- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files) +- **Search Tools** — Search provider configuration and testing +- **Memory** — Memory system management and visualization +- **Skills** — Skills framework management and execution +- **Translator** — Format debugging: playground, chat tester, test bench, live monitor +- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced +- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed) +- **Onboarding** — Setup wizard for new users +- **Usage** — Usage history and analytics +- **API Manager** — API key management with scoped permissions + +### Protocol Support +- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations` +- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens` +- **OpenAI Responses** — `/v1/responses` +- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}` +- **Ollama** — `/v1/api/chat`, `/api/tags` +- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily) +- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP) +- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills) +- **ACP** — Agent Communication Protocol registry and manager + +### MCP Server (25 Tools) +| Category | Tools | +|-----------|-------| +| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` | +| Memory (3) | `memory_search`, `memory_add`, `memory_clear` | +| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` | + +**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience` + +### Provider Categories + +**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI + +**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline + +**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan + +**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs + +### Internationalization +- 30 languages for UI (all dashboard pages) +- 30 translated documentation sets in docs/i18n/ +- Language switcher in documentation + +## Key Architectural Decisions + +1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints. + +2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently. + +3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. + +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. + +5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. + +6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes. + +7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`). + +8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. + +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations. + +10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. + +11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation. + +12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline. + +13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast. + +## Main Flows + +### Proxy Request Flow +1. Client sends OpenAI-format request to `/v1/chat/completions` +2. API key validation +3. Model resolution: direct model or combo lookup +4. For combos: iterate through models with selected strategy +5. Auth resolution: get credentials for the target provider +6. Format translation: OpenAI → provider native format +7. CLI fingerprint matching (if enabled for provider) +8. Upstream request with circuit breaker and rate limiting +9. Response translation: provider → OpenAI format +10. omniModel tag sanitization (strip internal tags) +11. SSE streaming back to client +12. Memory extraction (if memory system enabled) +13. Usage logging and cost calculation + +### OAuth Flow +1. Dashboard initiates `/api/oauth/[provider]/authorize` +2. User completes OAuth login in browser +3. Callback hits `/api/oauth/[provider]/exchange` +4. Tokens stored as a provider connection in SQLite +5. Background job refreshes tokens before expiry + +## Important Notes for LLMs + +1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only). + +2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`). + +3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services. + +4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. + +5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. + +6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). + +7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes. + +8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB. + +9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown. + +10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130). + +11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux). + +12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint. + +13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API. + +14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API. + +15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. + +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + +## Links + +- Repository: https://github.com/diegosouzapw/OmniRoute +- Website: https://omniroute.online +- npm: https://www.npmjs.com/package/omniroute +- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute +- Documentation: See `/docs/` directory diff --git a/llm.txt b/llm.txt index 3ca65dd096..c071f606d8 100644 --- a/llm.txt +++ b/llm.txt @@ -8,7 +8,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo **Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost. -**Current version:** 3.5.4 +**Current version:** 3.5.5 ## Tech Stack @@ -41,7 +41,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── auto-combo/ # Auto-combo engine dashboard │ │ │ ├── cache/ # Cache dashboard (semantic cache stats) │ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.) -│ │ │ ├── combos/ # Model combo management (9 strategies + 4 templates) +│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates) │ │ │ ├── costs/ # Cost tracking per provider/model │ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs │ │ │ ├── health/ # System health (uptime, circuit breakers, latency) @@ -238,7 +238,9 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ ├── ipFilter.ts # IP-based access control │ │ ├── signatureCache.ts # CLI signature caching │ │ ├── volumeDetector.ts # Request volume detection -│ │ └── ... # Additional services (16 more modules) +│ │ ├── contextHandoff.ts # Context relay handoff generation and injection +│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay +│ │ └── ... # Additional services (14 more modules) │ ├── transformer/ # Responses API transformer │ │ └── responsesTransformer.ts │ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek) @@ -277,13 +279,14 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo └── .env.example # Environment variable template ``` -## Key Features (v3.5.4) +## Key Features (v3.5.5) ### Core Proxy - **60+ AI providers** with automatic format translation - **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible) -- **9 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random +- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay - **4-tier fallback**: Subscription → API Key → Cheap → Free +- **Context Relay strategy**: Session handoff summaries on account rotation for continuity - **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown - **Semantic caching** with cache hit/miss headers - **Idempotency** with configurable dedup window @@ -299,7 +302,8 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches) ### Security -- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection) +- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs) +- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()` - **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()` - **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams - **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection @@ -310,7 +314,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ### Dashboard Pages (23 sections) - **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons -- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 9 strategies +- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies - **Auto-Combo** — Auto-combo engine dashboard with scoring metrics - **Analytics** — Token consumption, cost, heatmaps, distributions - **Health** — Uptime, memory, latency percentiles, circuit breakers @@ -376,7 +380,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo 3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation. -4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 9 strategies including auto-combo with self-healing. +4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity. 5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client. @@ -452,6 +456,12 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo 15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time. +16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`. + +17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22. + +18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false. + ## Links - Repository: https://github.com/diegosouzapw/OmniRoute diff --git a/open-sse/mcp-server/schemas/tools.ts b/open-sse/mcp-server/schemas/tools.ts index 30ca659b5b..99d15a20ce 100644 --- a/open-sse/mcp-server/schemas/tools.ts +++ b/open-sse/mcp-server/schemas/tools.ts @@ -113,6 +113,7 @@ export const listCombosOutput = z.object({ "priority", "weighted", "round-robin", + "context-relay", "strict-random", "random", "least-used", @@ -538,6 +539,7 @@ export const setRoutingStrategyInput = z.object({ "priority", "weighted", "round-robin", + "context-relay", "strict-random", "random", "least-used", diff --git a/open-sse/mcp-server/tools/advancedTools.ts b/open-sse/mcp-server/tools/advancedTools.ts index 8b74671b85..781e50ac68 100644 --- a/open-sse/mcp-server/tools/advancedTools.ts +++ b/open-sse/mcp-server/tools/advancedTools.ts @@ -343,6 +343,7 @@ export async function handleSetRoutingStrategy(args: { | "priority" | "weighted" | "round-robin" + | "context-relay" | "strict-random" | "random" | "least-used" diff --git a/open-sse/services/combo.ts b/open-sse/services/combo.ts index 04c8771e86..3388d0638a 100644 --- a/open-sse/services/combo.ts +++ b/open-sse/services/combo.ts @@ -1,12 +1,15 @@ /** * Shared combo (model combo) handling with fallback support - * Supports: priority, weighted, round-robin, random, least-used, and cost-optimized strategies + * Supports: priority, weighted, round-robin, random, least-used, cost-optimized, + * strict-random, auto, fill-first, p2c, lkgp, context-optimized, and context-relay strategies */ import { checkFallbackError, formatRetryAfter, getProviderProfile } from "./accountFallback.ts"; import { unavailableResponse } from "../utils/error.ts"; import { recordComboIntent, recordComboRequest, getComboMetrics } from "./comboMetrics.ts"; import { resolveComboConfig, getDefaultComboConfig } from "./comboConfig.ts"; +import { maybeGenerateHandoff, resolveContextRelayConfig } from "./contextHandoff.ts"; +import { fetchCodexQuota } from "./codexQuotaFetcher.ts"; import * as semaphore from "./rateLimitSemaphore.ts"; import { getCircuitBreaker } from "../../src/shared/utils/circuitBreaker"; import { fisherYatesShuffle, getNextFromDeck } from "../../src/shared/utils/shuffleDeck"; @@ -17,6 +20,7 @@ import { selectProvider as selectAutoProvider } from "./autoCombo/engine.ts"; import { selectWithStrategy } from "./autoCombo/routerStrategy.ts"; import { DEFAULT_WEIGHTS, scorePool } from "./autoCombo/scoring.ts"; import { supportsToolCalling } from "./modelCapabilities.ts"; +import { getSessionConnection } from "./sessionManager.ts"; import { getModelContextLimit } from "../../src/lib/modelsDevSync"; // Status codes that should mark semaphore + record circuit breaker failures @@ -523,8 +527,7 @@ async function buildAutoCandidates(modelStrings, comboName) { } /** - * Handle combo chat with fallback - * Supports all 6 strategies: priority, weighted, round-robin, random, least-used, cost-optimized + * Handle combo chat with fallback. * @param {Object} options * @param {Object} options.body - Request body * @param {Object} options.combo - Full combo object { name, models, strategy, config } @@ -542,9 +545,12 @@ export async function handleComboChat({ log, settings, allCombos, + relayOptions, }) { const strategy = combo.strategy || "priority"; const models = combo.models || []; + const relayConfig = + strategy === "context-relay" ? resolveContextRelayConfig(relayOptions?.config || null) : null; // ── Combo Agent Middleware (#399 + #401) ──────────────────────────────── // Apply system_message override, tool_filter_regex, and extract pinned model @@ -937,7 +943,6 @@ export async function handleComboChat({ let earliestRetryAfter = null; let lastStatus = null; const startTime = Date.now(); - let resolvedByModel = null; let fallbackCount = 0; for (let i = 0; i < orderedModels.length; i++) { @@ -1003,7 +1008,6 @@ export async function handleComboChat({ if (i > 0) fallbackCount++; break; // move to next model } - resolvedByModel = modelStr; const latencyMs = Date.now() - startTime; log.info( "COMBO", @@ -1017,6 +1021,39 @@ export async function handleComboChat({ strategy, }); + // Context-relay intentionally splits responsibilities: + // combo.ts decides whether a successful turn should generate a handoff, + // while chat.ts injects the handoff after the real connectionId is resolved. + if ( + strategy === "context-relay" && + relayOptions?.sessionId && + relayConfig && + relayConfig.handoffProviders.includes(provider) && + provider === "codex" + ) { + const connectionId = getSessionConnection(relayOptions.sessionId); + if (connectionId) { + const quotaInfo = await fetchCodexQuota(connectionId).catch(() => null); + if (quotaInfo) { + const resetCandidates = [quotaInfo.window5h?.resetAt, quotaInfo.window7d?.resetAt] + .filter((value): value is string => typeof value === "string" && value.length > 0) + .sort(); + + maybeGenerateHandoff({ + sessionId: relayOptions.sessionId, + comboName: combo.name, + connectionId, + percentUsed: quotaInfo.percentUsed, + messages: Array.isArray(body?.messages) ? body.messages : [], + model: modelStr, + expiresAt: resetCandidates[0] || null, + config: relayConfig, + handleSingleModel: handleSingleModelWrapped, + }); + } + } + } + // Record last known good provider (LKGP) for this combo/model (#919) if (provider) { import("../../src/lib/localDb") diff --git a/open-sse/services/comboConfig.ts b/open-sse/services/comboConfig.ts index 5eebabf22f..2b3269fd58 100644 --- a/open-sse/services/comboConfig.ts +++ b/open-sse/services/comboConfig.ts @@ -14,6 +14,10 @@ const DEFAULT_COMBO_CONFIG = { queueTimeoutMs: 30000, // max wait time in semaphore queue (round-robin) healthCheckEnabled: true, healthCheckTimeoutMs: 3000, + handoffThreshold: 0.85, + handoffModel: "", + handoffProviders: ["codex"], + maxMessagesForSummary: 30, maxComboDepth: 3, trackMetrics: true, }; diff --git a/open-sse/services/contextHandoff.ts b/open-sse/services/contextHandoff.ts new file mode 100644 index 0000000000..61e9875e7e --- /dev/null +++ b/open-sse/services/contextHandoff.ts @@ -0,0 +1,383 @@ +import { + cleanupExpiredHandoffs, + hasActiveHandoff, + type HandoffPayload, + upsertHandoff, +} from "../../src/lib/db/contextHandoffs.ts"; +import { estimateTokens } from "./contextManager.ts"; +import { stripMarkdownCodeFence } from "../utils/aiSdkCompat.ts"; + +export const HANDOFF_WARNING_THRESHOLD = 0.85; +export const HANDOFF_EXHAUSTION_THRESHOLD = 0.95; + +const MAX_HISTORY_TOKENS_FOR_SUMMARY = 8000; +const DEFAULT_MAX_MESSAGES_FOR_SUMMARY = 30; +const DEFAULT_SUMMARY_RESPONSE_TOKENS = 800; +const MAX_SUMMARY_LENGTH = 2000; +const MAX_TASK_PROGRESS_LENGTH = 1200; +const MAX_DECISIONS = 8; +const MAX_ENTITIES = 10; +const DEFAULT_TTL_MS = 5 * 60 * 60 * 1000; +const OMNI_MODEL_TAG_PATTERN = /(?:\\n|\n)?[^<]+<\/omniModel>(?:\\n|\n)?/g; +const inflightHandoffGenerations = new Set(); + +const HANDOFF_PROMPT_TEMPLATE = `You are a context summarizer. Analyze the conversation below and generate a structured handoff summary. +This summary will be used to restore context when this conversation is moved to a new AI account. + +CONVERSATION HISTORY: +{HISTORY} + +Generate a JSON object with this exact structure: +{ + "summary": "A clear, dense summary of what has been discussed and accomplished (max 200 words). Focus on what the AI needs to know to continue seamlessly.", + "keyDecisions": ["decision1", "decision2"], + "taskProgress": "Current state of the task: what's done, what's pending, next steps", + "activeEntities": ["file1.ts", "feature X", "topic Y"] +} + +Important: Return ONLY the JSON object, no markdown, no explanation.`; + +type MessageLike = { + role?: string; + content?: unknown; +}; + +export interface ContextRelayConfig { + handoffModel?: string; + handoffThreshold?: number; + handoffProviders?: string[]; + maxMessagesForSummary?: number; +} + +export interface ParsedHandoffContent { + summary: string; + keyDecisions: string[]; + taskProgress: string; + activeEntities: string[]; +} + +export function resolveContextRelayConfig( + config?: Record | null +): Required { + const rawThreshold = Number(config?.handoffThreshold); + const rawMaxMessages = Number(config?.maxMessagesForSummary); + const hasExplicitProviders = Array.isArray(config?.handoffProviders); + const handoffProviders = hasExplicitProviders + ? config?.handoffProviders + .map((item) => (typeof item === "string" ? item.trim().toLowerCase() : "")) + .filter(Boolean) + : ["codex"]; + + return { + handoffModel: + typeof config?.handoffModel === "string" && config.handoffModel.trim().length > 0 + ? config.handoffModel.trim() + : "", + handoffThreshold: + Number.isFinite(rawThreshold) && + rawThreshold > 0 && + rawThreshold < HANDOFF_EXHAUSTION_THRESHOLD + ? rawThreshold + : HANDOFF_WARNING_THRESHOLD, + handoffProviders: hasExplicitProviders ? handoffProviders : ["codex"], + maxMessagesForSummary: + Number.isFinite(rawMaxMessages) && rawMaxMessages >= 5 && rawMaxMessages <= 100 + ? Math.round(rawMaxMessages) + : DEFAULT_MAX_MESSAGES_FOR_SUMMARY, + }; +} + +function getInflightKey(sessionId: string, comboName: string): string { + return `${sessionId}::${comboName}`; +} + +function toTextContent(content: unknown): string { + if (typeof content === "string") return content; + if (!Array.isArray(content)) return ""; + + return content + .map((part) => { + if (!part || typeof part !== "object") return ""; + if (typeof (part as Record).text === "string") { + return String((part as Record).text); + } + if (typeof (part as Record).content === "string") { + return String((part as Record).content); + } + return ""; + }) + .filter(Boolean) + .join("\n"); +} + +function formatMessagesForPrompt(messages: MessageLike[]): string { + return messages + .map((message, index) => { + const role = typeof message.role === "string" ? message.role : "unknown"; + const content = toTextContent(message.content).trim(); + if (!content) return ""; + return `[${index + 1}] ${role.toUpperCase()}:\n${content}`; + }) + .filter(Boolean) + .join("\n\n"); +} + +function selectMessagesForSummary(messages: MessageLike[], maxMessages: number): MessageLike[] { + const recentMessages = messages.slice(-maxMessages); + let working = [...recentMessages]; + + while (working.length > 1) { + const history = formatMessagesForPrompt(working); + if (estimateTokens(history) <= MAX_HISTORY_TOKENS_FOR_SUMMARY) { + return working; + } + working = working.slice(1); + } + + return working; +} + +function normalizeStringArray(value: unknown, maxItems: number, maxLength = 240): string[] { + if (!Array.isArray(value)) return []; + return value + .map((item) => (typeof item === "string" ? item.trim() : "")) + .filter(Boolean) + .slice(0, maxItems) + .map((item) => item.slice(0, maxLength)); +} + +function sanitizeJsonCandidate(content: string): string { + return content.replace(OMNI_MODEL_TAG_PATTERN, "").trim(); +} + +function extractJsonCandidate(content: string): string { + const stripped = sanitizeJsonCandidate(String(stripMarkdownCodeFence(content) || "")); + if (!stripped) return ""; + + try { + JSON.parse(stripped); + return stripped; + } catch { + const firstBrace = stripped.indexOf("{"); + const lastBrace = stripped.lastIndexOf("}"); + if (firstBrace >= 0 && lastBrace > firstBrace) { + return stripped.slice(firstBrace, lastBrace + 1); + } + return stripped; + } +} + +export function parseHandoffJSON(content: string): ParsedHandoffContent | null { + const candidate = extractJsonCandidate(content); + if (!candidate) return null; + + try { + const parsed = JSON.parse(candidate) as Record; + const summary = + typeof parsed.summary === "string" ? parsed.summary.trim().slice(0, MAX_SUMMARY_LENGTH) : ""; + const taskProgress = + typeof parsed.taskProgress === "string" + ? parsed.taskProgress.trim().slice(0, MAX_TASK_PROGRESS_LENGTH) + : ""; + const keyDecisions = normalizeStringArray(parsed.keyDecisions, MAX_DECISIONS); + const activeEntities = normalizeStringArray(parsed.activeEntities, MAX_ENTITIES); + + if (!summary) return null; + + return { + summary, + keyDecisions, + taskProgress, + activeEntities, + }; + } catch { + return null; + } +} + +function escapeXml(value: string): string { + return value + .replaceAll("&", "&") + .replaceAll("<", "<") + .replaceAll(">", ">") + .replaceAll('"', """) + .replaceAll("'", "'"); +} + +function getResponseText(json: Record): string { + const choices = Array.isArray(json.choices) ? json.choices : []; + const firstChoice = choices[0] as Record | undefined; + const firstMessage = firstChoice?.message as Record | undefined; + if (typeof firstMessage?.content === "string") { + return firstMessage.content; + } + + if (Array.isArray(firstMessage?.content)) { + return toTextContent(firstMessage.content); + } + + const output = Array.isArray(json.output) ? json.output : []; + for (const item of output) { + if (!item || typeof item !== "object") continue; + const content = Array.isArray((item as Record).content) + ? ((item as Record).content as Array>) + : []; + for (const part of content) { + if (typeof part?.text === "string") return part.text; + } + } + + const content = Array.isArray(json.content) ? json.content : []; + for (const part of content) { + if (!part || typeof part !== "object") continue; + if (typeof (part as Record).text === "string") { + return String((part as Record).text); + } + } + + return ""; +} + +async function generateHandoffAsync(options: { + sessionId: string; + comboName: string; + connectionId: string; + percentUsed: number; + messages: MessageLike[]; + model: string; + expiresAt: string | null; + config?: ContextRelayConfig | null; + handleSingleModel: (body: Record, modelStr: string) => Promise; +}): Promise { + cleanupExpiredHandoffs(); + + const relayConfig = resolveContextRelayConfig(options.config as Record); + const summaryModel = relayConfig.handoffModel || options.model; + const selectedMessages = selectMessagesForSummary( + Array.isArray(options.messages) ? options.messages : [], + relayConfig.maxMessagesForSummary + ); + const historyText = formatMessagesForPrompt(selectedMessages); + if (!historyText) return; + + const summaryPrompt = HANDOFF_PROMPT_TEMPLATE.replace("{HISTORY}", historyText); + const summaryBody = { + model: summaryModel, + messages: [{ role: "user", content: summaryPrompt }], + stream: false, + max_tokens: DEFAULT_SUMMARY_RESPONSE_TOKENS, + temperature: 0.1, + _omnirouteSkipContextRelay: true, + _omnirouteInternalRequest: "context-handoff", + }; + + const response = await options.handleSingleModel(summaryBody, summaryModel); + if (!response.ok) return; + + let content = ""; + try { + const json = (await response.clone().json()) as Record; + content = getResponseText(json); + } catch { + try { + content = await response.clone().text(); + } catch { + content = ""; + } + } + + const parsed = parseHandoffJSON(content); + if (!parsed) return; + + upsertHandoff({ + sessionId: options.sessionId, + comboName: options.comboName, + fromAccount: options.connectionId, + summary: parsed.summary, + keyDecisions: parsed.keyDecisions, + taskProgress: parsed.taskProgress, + activeEntities: parsed.activeEntities, + messageCount: Array.isArray(options.messages) ? options.messages.length : 0, + model: summaryModel, + warningThresholdPct: relayConfig.handoffThreshold, + generatedAt: new Date().toISOString(), + expiresAt: options.expiresAt || new Date(Date.now() + DEFAULT_TTL_MS).toISOString(), + }); +} + +export function maybeGenerateHandoff(options: { + sessionId: string | null; + comboName: string; + connectionId: string | null; + percentUsed: number; + messages: MessageLike[]; + model: string; + expiresAt: string | null; + config?: ContextRelayConfig | null; + handleSingleModel: (body: Record, modelStr: string) => Promise; +}): void { + if (!options.sessionId || !options.connectionId) return; + + const relayConfig = resolveContextRelayConfig(options.config as Record); + if (relayConfig.handoffProviders.length === 0) return; + if (options.percentUsed < relayConfig.handoffThreshold) return; + if (options.percentUsed >= HANDOFF_EXHAUSTION_THRESHOLD) return; + + cleanupExpiredHandoffs(); + if (hasActiveHandoff(options.sessionId, options.comboName)) return; + const inflightKey = getInflightKey(options.sessionId, options.comboName); + if (inflightHandoffGenerations.has(inflightKey)) return; + inflightHandoffGenerations.add(inflightKey); + + setImmediate(() => { + generateHandoffAsync({ + ...options, + sessionId: options.sessionId as string, + connectionId: options.connectionId as string, + config: relayConfig, + }) + .catch((err) => { + if (process.env.NODE_ENV !== "test") { + console.warn("[context-relay] Handoff generation failed:", err?.message || err); + } + }) + .finally(() => { + inflightHandoffGenerations.delete(inflightKey); + }); + }); +} + +export function buildHandoffSystemMessage(payload: HandoffPayload): string { + const decisions = payload.keyDecisions.map((decision) => ` - ${escapeXml(decision)}`).join("\n"); + const entities = payload.activeEntities.map((entity) => escapeXml(entity)).join(", "); + + return ` +Account quota transfer - continuing from previous session +${escapeXml(payload.summary)} +${escapeXml(payload.taskProgress)} + +${decisions} + +${entities} +${payload.messageCount} + + +You are continuing a conversation that was transferred from another account due to quota limits. +The context above contains a concise summary of the prior work. Continue seamlessly from where the session left off.`; +} + +export function injectHandoffIntoBody( + body: Record, + payload: HandoffPayload +): Record { + const handoffMessage = { + role: "system", + content: buildHandoffSystemMessage(payload), + }; + const messages = Array.isArray(body.messages) ? [...body.messages] : []; + + return { + ...body, + messages: [handoffMessage, ...messages], + }; +} diff --git a/src/app/(dashboard)/dashboard/combos/page.tsx b/src/app/(dashboard)/dashboard/combos/page.tsx index 3c3845ba63..17f278abc3 100644 --- a/src/app/(dashboard)/dashboard/combos/page.tsx +++ b/src/app/(dashboard)/dashboard/combos/page.tsx @@ -29,6 +29,15 @@ const STRATEGY_OPTIONS = ROUTING_STRATEGIES.map((strategy) => ({ icon: strategy.icon, })); +const STRATEGY_LABEL_FALLBACK = { + "context-relay": "Context Relay", +}; + +const STRATEGY_DESC_FALLBACK = { + "context-relay": + "Priority-style routing with automatic context handoffs when account rotation happens.", +}; + const STRATEGY_GUIDANCE_FALLBACK = { priority: { when: "Use when you have one preferred model and only want fallback on failure.", @@ -45,6 +54,12 @@ const STRATEGY_GUIDANCE_FALLBACK = { avoid: "Avoid when model latency/cost differs significantly.", example: "Example: Same model across multiple accounts to spread throughput.", }, + "context-relay": { + when: "Use when long sessions must survive account rotation without losing the working context.", + avoid: + "Avoid when account switching is rare or when you do not want extra summarization requests.", + example: "Example: Codex sessions that rotate across multiple accounts near quota exhaustion.", + }, random: { when: "Use when you want a simple spread with low configuration effort.", avoid: "Avoid when requests must be distributed with strict guarantees.", @@ -114,6 +129,16 @@ const STRATEGY_RECOMMENDATIONS_FALLBACK = { "Use queue timeout to fail fast under saturation.", ], }, + "context-relay": { + title: "Session continuity first", + description: + "Best when account rotation is expected and the next account must inherit a condensed task summary.", + tips: [ + "Use with providers that rotate accounts for the same model family.", + "Keep the handoff threshold below the hard quota cutoff to give the summary time to generate.", + "Set a dedicated summary model only when the primary model is too expensive or unstable.", + ], + }, random: { title: "Quick spread with low setup", description: "Use when you need simple distribution without strict guarantees.", @@ -275,16 +300,24 @@ function getStrategyMeta(strategy) { } function getStrategyLabel(t, strategy) { - return t(getStrategyMeta(strategy).labelKey); + const key = getStrategyMeta(strategy).labelKey; + return getI18nOrFallback(t, key, STRATEGY_LABEL_FALLBACK[strategy] || strategy); } function getStrategyDescription(t, strategy) { - return t(getStrategyMeta(strategy).descKey); + const key = getStrategyMeta(strategy).descKey; + return getI18nOrFallback( + t, + key, + STRATEGY_DESC_FALLBACK[strategy] || STRATEGY_DESC_FALLBACK.priority || strategy + ); } function getStrategyBadgeClass(strategy) { if (strategy === "weighted") return "bg-amber-500/15 text-amber-600 dark:text-amber-400"; if (strategy === "round-robin") return "bg-emerald-500/15 text-emerald-600 dark:text-emerald-400"; + if (strategy === "context-relay") + return "bg-fuchsia-500/15 text-fuchsia-600 dark:text-fuchsia-400"; if (strategy === "random") return "bg-purple-500/15 text-purple-600 dark:text-purple-400"; if (strategy === "least-used") return "bg-cyan-500/15 text-cyan-600 dark:text-cyan-400"; if (strategy === "cost-optimized") return "bg-teal-500/15 text-teal-600 dark:text-teal-400"; @@ -1202,6 +1235,31 @@ function ComboFormModal({ isOpen, combo, onClose, onSave, activeProviders }) { !!combo?.context_cache_protection ); + const resetFormForCombo = useCallback( + (nextCombo, comboDefaults = null) => { + const nextDefaults = + nextCombo || comboDefaults + ? { + ...(comboDefaults || {}), + } + : {}; + const nextConfig = nextCombo?.config + ? { ...nextCombo.config } + : Object.fromEntries(Object.entries(nextDefaults).filter(([key]) => key !== "strategy")); + + setName(nextCombo?.name || ""); + setModels((nextCombo?.models || []).map((m) => normalizeModelEntry(m))); + setStrategy(nextCombo?.strategy || comboDefaults?.strategy || "priority"); + setConfig(nextConfig); + setShowAdvanced(false); + setNameError(""); + setAgentSystemMessage(nextCombo?.system_message || ""); + setAgentToolFilter(nextCombo?.tool_filter_regex || ""); + setAgentContextCache(!!nextCombo?.context_cache_protection); + }, + [setAgentContextCache] + ); + // DnD state const hasPricingForModel = useCallback( (modelValue) => { @@ -1335,6 +1393,39 @@ function ComboFormModal({ isOpen, combo, onClose, onSave, activeProviders }) { if (isOpen) fetchModalData(); }, [isOpen]); + useEffect(() => { + if (!isOpen) return; + + let cancelled = false; + + if (combo) { + resetFormForCombo(combo); + return () => { + cancelled = true; + }; + } + + const loadDefaults = async () => { + try { + const response = await fetch("/api/settings/combo-defaults"); + const data = response.ok ? await response.json() : {}; + if (!cancelled) { + resetFormForCombo(null, data.comboDefaults || null); + } + } catch { + if (!cancelled) { + resetFormForCombo(null, null); + } + } + }; + + loadDefaults(); + + return () => { + cancelled = true; + }; + }, [combo, isOpen, resetFormForCombo]); + useEffect(() => { if (!strategyChangeMountedRef.current) { strategyChangeMountedRef.current = true; @@ -1409,6 +1500,13 @@ function ComboFormModal({ isOpen, combo, onClose, onSave, activeProviders }) { concurrencyPerModel: 3, queueTimeoutMs: 30000, }, + "context-relay": { + maxRetries: 1, + retryDelayMs: 750, + healthCheckEnabled: true, + handoffThreshold: 0.85, + maxMessagesForSummary: 30, + }, random: { maxRetries: 1, retryDelayMs: 1000, healthCheckEnabled: true }, "least-used": { maxRetries: 1, retryDelayMs: 1000, healthCheckEnabled: true }, "cost-optimized": { maxRetries: 1, retryDelayMs: 500, healthCheckEnabled: true }, @@ -1667,8 +1765,11 @@ function ComboFormModal({ isOpen, combo, onClose, onSave, activeProviders }) { key={s.value} onClick={() => setStrategy(s.value)} data-testid={`strategy-option-${s.value}`} - title={t(s.descKey)} - aria-label={`${getStrategyLabel(t, s.value)}. ${t(s.descKey)}`} + title={getStrategyDescription(t, s.value)} + aria-label={`${getStrategyLabel(t, s.value)}. ${getStrategyDescription( + t, + s.value + )}`} className={`py-1.5 px-2 rounded-md text-xs font-medium transition-all ${ strategy === s.value ? "bg-white dark:bg-bg-main shadow-sm text-primary" @@ -2085,6 +2186,100 @@ function ComboFormModal({ isOpen, combo, onClose, onSave, activeProviders }) {
)} + {strategy === "context-relay" && ( +
+
+ + + setConfig({ + ...config, + handoffThreshold: e.target.value ? Number(e.target.value) : undefined, + }) + } + className="w-full text-xs py-1.5 px-2 rounded border border-black/10 dark:border-white/10 bg-transparent focus:border-primary focus:outline-none" + /> +
+
+ + + setConfig({ + ...config, + maxMessagesForSummary: e.target.value + ? Number(e.target.value) + : undefined, + }) + } + className="w-full text-xs py-1.5 px-2 rounded border border-black/10 dark:border-white/10 bg-transparent focus:border-primary focus:outline-none" + /> +
+
+ + + setConfig({ + ...config, + handoffModel: e.target.value || undefined, + }) + } + className="w-full text-xs py-1.5 px-2 rounded border border-black/10 dark:border-white/10 bg-transparent focus:border-primary focus:outline-none" + /> +
+
+

+ {getI18nOrFallback( + t, + "contextRelayProviderNote", + "Context Relay currently generates handoffs for Codex account rotation. Pair it with multiple accounts of the same provider for the best continuity." + )} +

+
+
+ )}

{t("advancedHint")}

)} diff --git a/src/app/(dashboard)/dashboard/settings/components/ComboDefaultsTab.tsx b/src/app/(dashboard)/dashboard/settings/components/ComboDefaultsTab.tsx index fab596e912..222b6820f6 100644 --- a/src/app/(dashboard)/dashboard/settings/components/ComboDefaultsTab.tsx +++ b/src/app/(dashboard)/dashboard/settings/components/ComboDefaultsTab.tsx @@ -6,6 +6,18 @@ import { cn } from "@/shared/utils/cn"; import { ROUTING_STRATEGIES } from "@/shared/constants/routingStrategies"; import { useTranslations } from "next-intl"; +const STRATEGY_LABEL_FALLBACKS: Record = { + "context-relay": "Context Relay", +}; + +function translateOrFallback( + t: ReturnType, + key: string, + fallback: string +): string { + return typeof t.has === "function" && t.has(key) ? t(key) : fallback; +} + export default function ComboDefaultsTab() { const [comboDefaults, setComboDefaults] = useState({ strategy: "priority", @@ -16,6 +28,9 @@ export default function ComboDefaultsTab() { healthCheckTimeoutMs: 3000, maxComboDepth: 3, trackMetrics: true, + handoffThreshold: 0.85, + handoffModel: "", + maxMessagesForSummary: 30, }); const [providerOverrides, setProviderOverrides] = useState({}); const [newOverrideProvider, setNewOverrideProvider] = useState(""); @@ -24,7 +39,11 @@ export default function ComboDefaultsTab() { const tc = useTranslations("common"); const strategyOptions = ROUTING_STRATEGIES.map((strategy) => ({ value: strategy.value, - label: t(strategy.labelKey), + label: translateOrFallback( + t, + strategy.labelKey, + STRATEGY_LABEL_FALLBACKS[strategy.value] || strategy.value + ), icon: strategy.icon, })); const numericSettings = [ @@ -38,7 +57,9 @@ export default function ComboDefaultsTab() { fetch("/api/settings/combo-defaults") .then((res) => res.json()) .then((data) => { - if (data.comboDefaults) setComboDefaults(data.comboDefaults); + if (data.comboDefaults) { + setComboDefaults((prev) => ({ ...prev, ...data.comboDefaults })); + } if (data.providerOverrides) setProviderOverrides(data.providerOverrides); }) .catch((err) => console.error("Failed to fetch combo defaults:", err)); @@ -180,6 +201,64 @@ export default function ComboDefaultsTab() { )} + {comboDefaults.strategy === "context-relay" && ( +
+ + setComboDefaults((prev) => ({ + ...prev, + handoffThreshold: e.target.value ? Number(e.target.value) : undefined, + })) + } + className="text-sm" + /> + + setComboDefaults((prev) => ({ + ...prev, + maxMessagesForSummary: e.target.value ? Number(e.target.value) : undefined, + })) + } + className="text-sm" + /> + + setComboDefaults((prev) => ({ + ...prev, + handoffModel: e.target.value, + })) + } + className="text-sm" + /> +
+

+ {translateOrFallback( + t, + "contextRelayProviderNote", + "Context Relay currently generates handoffs for Codex accounts and uses these values as global defaults for new or unconfigured combos." + )} +

+
+
+ )} + {/* Toggles */}
diff --git a/src/app/api/settings/combo-defaults/route.ts b/src/app/api/settings/combo-defaults/route.ts index bcfafa6a27..fb2df51cda 100644 --- a/src/app/api/settings/combo-defaults/route.ts +++ b/src/app/api/settings/combo-defaults/route.ts @@ -18,6 +18,9 @@ export async function GET() { timeoutMs: 120000, healthCheckEnabled: true, healthCheckTimeoutMs: 3000, + handoffThreshold: 0.85, + handoffModel: "", + maxMessagesForSummary: 30, maxComboDepth: 3, trackMetrics: true, }, diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index bb36fe5ecb..c512b3a4c4 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -833,6 +833,8 @@ "priorityDesc": "Sequential fallback: tries model 1 first, then 2, etc.", "weightedDesc": "Distributes traffic by weight percentage with fallback", "roundRobinDesc": "Circular distribution: each request goes to the next model in rotation", + "contextRelay": "Context Relay", + "contextRelayDesc": "Preserves session continuity with handoff summaries when accounts rotate", "randomDesc": "Uniform random selection, then fallback to remaining models", "leastUsedDesc": "Picks the model with fewest requests, balancing load over time", "costOptimizedDesc": "Routes to the cheapest model first based on pricing", @@ -844,6 +846,13 @@ "retryDelay": "Retry Delay (ms)", "concurrencyPerModel": "Concurrency / Model", "queueTimeout": "Queue Timeout (ms)", + "contextRelayHandoffThreshold": "Handoff Threshold", + "contextRelayHandoffThresholdHelp": "When quota usage reaches this threshold, OmniRoute generates a structured handoff summary before the active account is exhausted.", + "contextRelayMaxMessages": "Max Messages For Summary", + "contextRelayMaxMessagesHelp": "Limits how much recent history is condensed into the relay summary.", + "contextRelaySummaryModel": "Summary Model", + "contextRelaySummaryModelHelp": "Optional override model used only for generating the handoff summary. Leave empty to reuse the active combo model.", + "contextRelayProviderNote": "Context Relay currently generates handoffs for Codex account rotation. Pair it with multiple accounts of the same provider for the best continuity.", "advancedHint": "Leave empty to use global defaults. These override per-provider settings.", "moveUp": "Move up", "moveDown": "Move down", @@ -2119,6 +2128,8 @@ "comboStrategyAria": "Combo strategy", "priority": "Priority", "weighted": "Weighted", + "contextRelay": "Context Relay", + "contextRelayDesc": "Priority-style routing with automatic context handoffs when account rotation happens", "maxRetriesLabel": "Max Retries", "retryDelayLabel": "Retry Delay (ms)", "timeoutLabel": "Timeout (ms)", @@ -2139,6 +2150,10 @@ "maxNestingDepth": "Max Nesting Depth", "concurrencyPerModel": "Concurrency / Model", "queueTimeout": "Queue Timeout (ms)", + "contextRelayHandoffThreshold": "Handoff Threshold", + "contextRelayMaxMessages": "Max Messages For Summary", + "contextRelaySummaryModel": "Summary Model", + "contextRelayProviderNote": "Context Relay currently generates handoffs for Codex accounts and uses these values as global defaults for new or unconfigured combos.", "providerProfiles": "Provider Profiles", "providerProfilesDesc": "Separate resilience settings for OAuth (session-based) and API Key (metered) providers. OAuth providers have stricter thresholds due to lower rate limits.", "oauthProviders": "OAuth Providers", diff --git a/src/i18n/messages/pt-BR.json b/src/i18n/messages/pt-BR.json index 86b1d38249..c792e5dc2b 100644 --- a/src/i18n/messages/pt-BR.json +++ b/src/i18n/messages/pt-BR.json @@ -833,6 +833,8 @@ "priorityDesc": "Fallback sequencial: tenta modelo 1 primeiro, depois 2, etc.", "weightedDesc": "Distribui tráfego por porcentagem de peso com fallback", "roundRobinDesc": "Distribuição circular: cada requisição vai para o próximo modelo na rotação", + "contextRelay": "Context Relay", + "contextRelayDesc": "Preserva a continuidade da sessão com resumos de handoff quando as contas giram", "randomDesc": "Seleção aleatória uniforme, depois fallback para modelos restantes", "leastUsedDesc": "Escolhe o modelo com menos requisições, equilibrando carga ao longo do tempo", "costOptimizedDesc": "Roteia para o modelo mais barato primeiro baseado em preços", @@ -844,6 +846,13 @@ "retryDelay": "Intervalo de Tentativa (ms)", "concurrencyPerModel": "Concorrência / Modelo", "queueTimeout": "Timeout da Fila (ms)", + "contextRelayHandoffThreshold": "Limite de Handoff", + "contextRelayHandoffThresholdHelp": "Quando o uso de quota atinge este limite, o OmniRoute gera um resumo estruturado antes de a conta ativa esgotar.", + "contextRelayMaxMessages": "Máx. de Mensagens no Resumo", + "contextRelayMaxMessagesHelp": "Limita quanto histórico recente será condensado no resumo de relay.", + "contextRelaySummaryModel": "Modelo do Resumo", + "contextRelaySummaryModelHelp": "Modelo opcional usado apenas para gerar o resumo de handoff. Deixe vazio para reutilizar o modelo ativo do combo.", + "contextRelayProviderNote": "O Context Relay atualmente gera handoffs para rotação de contas Codex. Combine com múltiplas contas do mesmo provedor para melhor continuidade.", "advancedHint": "Deixe vazio para usar padrões globais. Estes substituem configurações por provedor.", "moveUp": "Mover para cima", "moveDown": "Mover para baixo", @@ -2073,6 +2082,8 @@ "comboStrategyAria": "Estratégia de combo", "priority": "Prioridade", "weighted": "Ponderado", + "contextRelay": "Context Relay", + "contextRelayDesc": "Roteamento estilo prioridade com handoffs automáticos de contexto quando a conta gira", "maxRetriesLabel": "Máx. Tentativas", "retryDelayLabel": "Atraso entre Tentativas (ms)", "timeoutLabel": "Timeout (ms)", @@ -2093,6 +2104,10 @@ "maxNestingDepth": "Profundidade Máx. de Aninhamento", "concurrencyPerModel": "Concorrência / Modelo", "queueTimeout": "Timeout da Fila (ms)", + "contextRelayHandoffThreshold": "Limite de Handoff", + "contextRelayMaxMessages": "Máx. de Mensagens no Resumo", + "contextRelaySummaryModel": "Modelo do Resumo", + "contextRelayProviderNote": "O Context Relay atualmente gera handoffs para contas Codex e usa estes valores como padrão global para combos novos ou não configurados.", "providerProfiles": "Perfis de Provedor", "providerProfilesDesc": "Configurações de resiliência separadas para provedores OAuth (baseados em sessão) e API Key (medidos). Provedores OAuth têm limites mais rigorosos devido a taxas mais baixas.", "oauthProviders": "Provedores OAuth", diff --git a/src/lib/db/contextHandoffs.ts b/src/lib/db/contextHandoffs.ts new file mode 100644 index 0000000000..89f7b07427 --- /dev/null +++ b/src/lib/db/contextHandoffs.ts @@ -0,0 +1,167 @@ +import { getDbInstance, rowToCamel } from "./core"; + +export interface HandoffPayload { + id?: string; + sessionId: string; + comboName: string; + fromAccount: string; + summary: string; + keyDecisions: string[]; + taskProgress: string; + activeEntities: string[]; + messageCount: number; + model: string; + warningThresholdPct: number; + generatedAt: string; + expiresAt: string; + createdAt?: string; +} + +type JsonRecord = Record; + +interface StatementLike { + get: (...params: unknown[]) => TRow | undefined; + run: (...params: unknown[]) => { changes: number }; +} + +interface DbLike { + prepare: (sql: string) => StatementLike; +} + +const CLEANUP_THROTTLE_MS = 30 * 60 * 1000; + +let lastCleanupAt = 0; + +function parseJsonArray(value: unknown): string[] { + if (Array.isArray(value)) { + return value.map((item) => (typeof item === "string" ? item : String(item))).filter(Boolean); + } + + if (typeof value !== "string" || value.trim().length === 0) { + return []; + } + + try { + const parsed = JSON.parse(value); + return Array.isArray(parsed) + ? parsed.map((item) => (typeof item === "string" ? item : String(item))).filter(Boolean) + : []; + } catch { + return []; + } +} + +function toHandoffPayload(row: unknown): HandoffPayload | null { + const camel = rowToCamel(row) as JsonRecord | null; + if (!camel) return null; + + return { + id: typeof camel.id === "string" ? camel.id : undefined, + sessionId: typeof camel.sessionId === "string" ? camel.sessionId : "", + comboName: typeof camel.comboName === "string" ? camel.comboName : "", + fromAccount: typeof camel.fromAccount === "string" ? camel.fromAccount : "", + summary: typeof camel.summary === "string" ? camel.summary : "", + keyDecisions: parseJsonArray(camel.keyDecisions), + taskProgress: typeof camel.taskProgress === "string" ? camel.taskProgress : "", + activeEntities: parseJsonArray(camel.activeEntities), + messageCount: Number.isFinite(Number(camel.messageCount)) ? Number(camel.messageCount) : 0, + model: typeof camel.model === "string" ? camel.model : "", + warningThresholdPct: Number.isFinite(Number(camel.warningThresholdPct)) + ? Number(camel.warningThresholdPct) + : 0.85, + generatedAt: typeof camel.generatedAt === "string" ? camel.generatedAt : "", + expiresAt: typeof camel.expiresAt === "string" ? camel.expiresAt : "", + createdAt: typeof camel.createdAt === "string" ? camel.createdAt : undefined, + }; +} + +export function upsertHandoff(payload: HandoffPayload): void { + const db = getDbInstance() as unknown as DbLike; + const createdAt = new Date().toISOString(); + + db.prepare( + `INSERT INTO context_handoffs + (session_id, combo_name, from_account, summary, key_decisions, + task_progress, active_entities, message_count, model, + warning_threshold_pct, generated_at, expires_at, created_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(session_id, combo_name) DO UPDATE SET + from_account = excluded.from_account, + summary = excluded.summary, + key_decisions = excluded.key_decisions, + task_progress = excluded.task_progress, + active_entities = excluded.active_entities, + message_count = excluded.message_count, + model = excluded.model, + warning_threshold_pct = excluded.warning_threshold_pct, + generated_at = excluded.generated_at, + expires_at = excluded.expires_at, + created_at = excluded.created_at` + ).run( + payload.sessionId, + payload.comboName, + payload.fromAccount, + payload.summary, + JSON.stringify(payload.keyDecisions || []), + payload.taskProgress, + JSON.stringify(payload.activeEntities || []), + payload.messageCount, + payload.model, + payload.warningThresholdPct, + payload.generatedAt, + payload.expiresAt, + createdAt + ); +} + +export function getHandoff(sessionId: string, comboName: string): HandoffPayload | null { + const db = getDbInstance() as unknown as DbLike; + const now = new Date().toISOString(); + const row = db + .prepare( + `SELECT * + FROM context_handoffs + WHERE session_id = ? AND combo_name = ? AND expires_at > ? + ORDER BY created_at DESC + LIMIT 1` + ) + .get(sessionId, comboName, now); + + return toHandoffPayload(row); +} + +export function deleteHandoff(sessionId: string, comboName: string): void { + const db = getDbInstance() as unknown as DbLike; + db.prepare("DELETE FROM context_handoffs WHERE session_id = ? AND combo_name = ?").run( + sessionId, + comboName + ); +} + +export function cleanupExpiredHandoffs(): number { + const nowMs = Date.now(); + if (nowMs - lastCleanupAt < CLEANUP_THROTTLE_MS) { + return 0; + } + + const db = getDbInstance() as unknown as DbLike; + const now = new Date(nowMs).toISOString(); + const result = db.prepare("DELETE FROM context_handoffs WHERE expires_at <= ?").run(now); + lastCleanupAt = nowMs; + return result.changes; +} + +export function hasActiveHandoff(sessionId: string, comboName: string): boolean { + const db = getDbInstance() as unknown as DbLike; + const now = new Date().toISOString(); + const row = db + .prepare( + `SELECT 1 + FROM context_handoffs + WHERE session_id = ? AND combo_name = ? AND expires_at > ? + LIMIT 1` + ) + .get(sessionId, comboName, now); + + return !!row; +} diff --git a/src/lib/db/migrations/019_context_handoffs.sql b/src/lib/db/migrations/019_context_handoffs.sql new file mode 100644 index 0000000000..86ec995a11 --- /dev/null +++ b/src/lib/db/migrations/019_context_handoffs.sql @@ -0,0 +1,28 @@ +-- Migration 019: Context handoffs for context-relay combo strategy +-- Stores structured LLM-generated summaries used to bridge account switches. + +CREATE TABLE IF NOT EXISTS context_handoffs ( + id TEXT PRIMARY KEY DEFAULT (lower(hex(randomblob(8)))), + session_id TEXT NOT NULL, + combo_name TEXT NOT NULL, + from_account TEXT NOT NULL, + summary TEXT NOT NULL, + key_decisions TEXT NOT NULL DEFAULT '[]', + task_progress TEXT NOT NULL DEFAULT '', + active_entities TEXT NOT NULL DEFAULT '[]', + message_count INTEGER NOT NULL DEFAULT 0, + model TEXT NOT NULL DEFAULT '', + warning_threshold_pct REAL NOT NULL DEFAULT 0.85, + generated_at TEXT NOT NULL, + expires_at TEXT NOT NULL, + created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ', 'now')) +); + +CREATE INDEX IF NOT EXISTS idx_context_handoffs_session + ON context_handoffs(session_id, expires_at); + +CREATE INDEX IF NOT EXISTS idx_context_handoffs_expires + ON context_handoffs(expires_at); + +CREATE UNIQUE INDEX IF NOT EXISTS idx_context_handoffs_session_combo + ON context_handoffs(session_id, combo_name); diff --git a/src/shared/constants/routingStrategies.ts b/src/shared/constants/routingStrategies.ts index b67908a074..e2ebcdffb3 100644 --- a/src/shared/constants/routingStrategies.ts +++ b/src/shared/constants/routingStrategies.ts @@ -2,6 +2,7 @@ export type RoutingStrategyValue = | "priority" | "weighted" | "round-robin" + | "context-relay" | "fill-first" | "p2c" | "random" @@ -42,6 +43,13 @@ export const ROUTING_STRATEGIES: RoutingStrategyOption[] = [ settingsDescKey: "roundRobinDesc", icon: "autorenew", }, + { + value: "context-relay", + labelKey: "contextRelay", + combosDescKey: "contextRelayDesc", + settingsDescKey: "contextRelayDesc", + icon: "sync_alt", + }, { value: "fill-first", labelKey: "fillFirst", diff --git a/src/shared/schemas/validation.ts b/src/shared/schemas/validation.ts index e73a657a38..784518a4d0 100644 --- a/src/shared/schemas/validation.ts +++ b/src/shared/schemas/validation.ts @@ -39,7 +39,15 @@ export const comboSchema = z.object({ model: z.string().min(1, "Model pattern is required"), endpoint: z.enum(["chat", "embeddings", "images"]).default("chat"), strategy: z - .enum(["priority", "weighted", "round-robin", "random", "least-used", "cost-optimized"]) + .enum([ + "priority", + "weighted", + "round-robin", + "context-relay", + "random", + "least-used", + "cost-optimized", + ]) .default("priority"), nodes: z.array(comboNodeSchema).min(1, "At least one node is required"), isActive: z.boolean().default(true), diff --git a/src/shared/validation/schemas.ts b/src/shared/validation/schemas.ts index 6ea2a1fd71..048a2615d9 100644 --- a/src/shared/validation/schemas.ts +++ b/src/shared/validation/schemas.ts @@ -72,20 +72,11 @@ const comboModelEntry = z.union([ }), ]); -// Per-combo config overrides -const comboConfigSchema = z - .object({ - maxRetries: z.number().int().min(0).max(10).optional(), - retryDelayMs: z.number().int().min(0).max(60000).optional(), - timeoutMs: z.number().int().min(1000).max(600000).optional(), - healthCheckEnabled: z.boolean().optional(), - }) - .optional(); - const comboStrategySchema = z.enum([ "priority", "weighted", "round-robin", + "context-relay", "random", "least-used", "cost-optimized", @@ -121,6 +112,10 @@ const comboRuntimeConfigSchema = z queueTimeoutMs: z.coerce.number().int().min(1000).max(120000).optional(), healthCheckEnabled: z.boolean().optional(), healthCheckTimeoutMs: z.coerce.number().int().min(100).max(30000).optional(), + handoffThreshold: z.coerce.number().min(0.5).max(0.94).optional(), + handoffModel: z.string().trim().max(200).optional(), + handoffProviders: z.array(z.string().trim().min(1).max(100)).max(10).optional(), + maxMessagesForSummary: z.coerce.number().int().min(5).max(100).optional(), maxComboDepth: z.coerce.number().int().min(1).max(10).optional(), trackMetrics: z.boolean().optional(), // Auto-Combo / LKGP Extensions @@ -141,7 +136,7 @@ export const createComboSchema = z.object({ .regex(/^[a-zA-Z0-9_/.-]+$/, "Name can only contain letters, numbers, -, _, / and ."), models: z.array(comboModelEntry).optional().default([]), strategy: comboStrategySchema.optional().default("priority"), - config: comboConfigSchema, + config: comboRuntimeConfigSchema.optional(), allowedProviders: z.array(z.string().max(200)).optional(), system_message: z.string().max(50000).optional(), tool_filter_regex: z.string().max(1000).optional(), diff --git a/src/sse/handlers/chat.ts b/src/sse/handlers/chat.ts index 619e5455d7..ab884ec456 100644 --- a/src/sse/handlers/chat.ts +++ b/src/sse/handlers/chat.ts @@ -8,9 +8,12 @@ import { import { getModelInfo, getComboForModel } from "../services/model"; import { errorResponse } from "@omniroute/open-sse/utils/error.ts"; import { handleComboChat } from "@omniroute/open-sse/services/combo.ts"; +import { resolveComboConfig } from "@omniroute/open-sse/services/comboConfig.ts"; +import { injectHandoffIntoBody } from "@omniroute/open-sse/services/contextHandoff.ts"; import { HTTP_STATUS } from "@omniroute/open-sse/config/constants.ts"; import * as log from "../utils/logger"; import { checkAndRefreshToken } from "../services/tokenRefresh"; +import { deleteHandoff, getHandoff } from "@/lib/db/contextHandoffs"; import { getSettings, getCombos } from "@/lib/localDb"; import { sanitizeRequest } from "../../shared/utils/inputSanitizer"; import { @@ -299,8 +302,12 @@ export async function handleChat(request: any, clientRawRequest: any = null) { getSettings().catch(() => ({})), getCombos().catch(() => []), ]); + const relayConfig = + combo.strategy === "context-relay" ? resolveComboConfig(combo, settings) : null; telemetry.endPhase(); + // Context-relay keeps generation in combo.ts, but handoff injection lives here + // because only this layer knows which connectionId was actually selected. const response = await (handleComboChat as any)({ body, combo, @@ -324,6 +331,13 @@ export async function handleChat(request: any, clientRawRequest: any = null) { log, settings, allCombos, + relayOptions: + combo.strategy === "context-relay" + ? { + sessionId, + config: relayConfig, + } + : undefined, }); // ── Global Fallback Provider (#689) ──────────────────────────────────── @@ -484,7 +498,31 @@ async function handleSingleModelChat( const accountId = credentials.connectionId.slice(0, 8); log.info("AUTH", `Using ${provider} account: ${accountId}...`); - if (runtimeOptions.sessionId) { + let requestBody = body; + let injectedHandoff = null; + if ( + comboStrategy === "context-relay" && + comboName && + runtimeOptions.sessionId && + body?._omnirouteSkipContextRelay !== true && + !excludeConnectionId + ) { + const handoff = getHandoff(runtimeOptions.sessionId, comboName); + if (handoff && handoff.fromAccount !== credentials.connectionId) { + // Inject only after a real account switch. The combo loop itself cannot + // reliably detect this because account selection happens inside auth. + requestBody = injectHandoffIntoBody(body, handoff); + injectedHandoff = handoff; + log.info( + "CONTEXT_RELAY", + `Injecting handoff for session ${runtimeOptions.sessionId}: ${handoff.fromAccount.slice( + 0, + 8 + )} -> ${credentials.connectionId.slice(0, 8)}` + ); + } + } + if (runtimeOptions.sessionId && body?._omnirouteInternalRequest !== "context-handoff") { touchSession(runtimeOptions.sessionId, credentials.connectionId); } @@ -497,7 +535,7 @@ async function handleSingleModelChat( const { result, tlsFingerprintUsed } = await executeChatWithBreaker({ bypassCircuitBreaker: forceLiveComboTest, breaker, - body, + body: requestBody, provider, model, refreshedCredentials, @@ -533,6 +571,9 @@ async function handleSingleModelChat( if (result.success) { clearModelUnavailability(provider, model); + if (injectedHandoff && runtimeOptions.sessionId && comboName) { + deleteHandoff(runtimeOptions.sessionId, comboName); + } if (telemetry) telemetry.startPhase("finalize"); if (telemetry) telemetry.endPhase(); return result.response; diff --git a/src/types/combo.ts b/src/types/combo.ts index f337a924c2..9fbfa87cd4 100644 --- a/src/types/combo.ts +++ b/src/types/combo.ts @@ -16,7 +16,7 @@ export interface Combo { updatedAt: string; } -export type ComboStrategy = "priority" | "weighted" | "round-robin"; +export type ComboStrategy = "priority" | "weighted" | "round-robin" | "context-relay"; export interface ComboNode { connectionId: string; diff --git a/src/types/settings.ts b/src/types/settings.ts index 15f84d3a23..6751e09bf3 100644 --- a/src/types/settings.ts +++ b/src/types/settings.ts @@ -21,7 +21,7 @@ export interface Settings { } export interface ComboDefaults { - strategy: "priority" | "weighted" | "round-robin"; + strategy: "priority" | "weighted" | "round-robin" | "context-relay"; maxRetries: number; retryDelayMs: number; timeoutMs: number; @@ -31,6 +31,10 @@ export interface ComboDefaults { trackMetrics: boolean; concurrencyPerModel?: number; queueTimeoutMs?: number; + handoffThreshold?: number; + handoffModel?: string; + handoffProviders?: string[]; + maxMessagesForSummary?: number; } export interface ProxyConfig { diff --git a/tests/unit/chat-context-relay.test.mjs b/tests/unit/chat-context-relay.test.mjs new file mode 100644 index 0000000000..ca86ccba80 --- /dev/null +++ b/tests/unit/chat-context-relay.test.mjs @@ -0,0 +1,197 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { createChatPipelineHarness } from "../integration/_chatPipelineHarness.mjs"; + +const harness = await createChatPipelineHarness("chat-context-relay"); +const { buildRequest, combosDb, handleChat, resetStorage, waitFor } = harness; +const providersDb = await import("../../src/lib/db/providers.ts"); +const handoffDb = await import("../../src/lib/db/contextHandoffs.ts"); + +function buildResponsesResponse(text = "ok", model = "gpt-5.4") { + return new Response( + JSON.stringify({ + id: "resp_context_relay", + object: "response", + model, + output: [ + { + type: "message", + role: "assistant", + content: [{ type: "output_text", text }], + }, + ], + usage: { + input_tokens: 4, + output_tokens: 2, + }, + }), + { + status: 200, + headers: { "content-type": "application/json" }, + } + ); +} + +async function seedCodexOAuthConnection({ + name, + email, + accessToken, + refreshToken, + workspaceId, + priority, +}) { + return providersDb.createProviderConnection({ + provider: "codex", + authType: "oauth", + name, + email, + accessToken, + refreshToken, + isActive: true, + testStatus: "active", + priority, + providerSpecificData: { workspaceId }, + }); +} + +function buildQuotaResponse(usedPercent, resetAfterSeconds = 3600) { + return new Response( + JSON.stringify({ + rate_limit: { + primary_window: { + used_percent: usedPercent, + reset_after_seconds: resetAfterSeconds, + }, + secondary_window: { + used_percent: 0, + reset_after_seconds: 86400, + }, + }, + }), + { + status: 200, + headers: { "content-type": "application/json" }, + } + ); +} + +test.beforeEach(async () => { + await resetStorage(); +}); + +test.after(async () => { + await harness.cleanup(); +}); + +test("handleChat generates and injects context-relay handoffs across Codex account switches", async () => { + const primary = await seedCodexOAuthConnection({ + name: "codex-a", + email: "relay-a@example.com", + accessToken: "token-a", + refreshToken: "refresh-a", + workspaceId: "ws-a", + priority: 1, + }); + await seedCodexOAuthConnection({ + name: "codex-b", + email: "relay-b@example.com", + accessToken: "token-b", + refreshToken: "refresh-b", + workspaceId: "ws-b", + priority: 2, + }); + + await combosDb.createCombo({ + name: "relay-combo", + strategy: "context-relay", + config: { + maxRetries: 0, + retryDelayMs: 0, + handoffThreshold: 0.85, + maxMessagesForSummary: 12, + }, + models: ["codex/gpt-5.4"], + }); + + const upstreamBodies = []; + const summaryBodies = []; + + globalThis.fetch = async (url, init = {}) => { + const urlStr = String(url); + const headers = Object.fromEntries(new Headers(init.headers || {}).entries()); + + if (urlStr.includes("/backend-api/wham/usage")) { + const authHeader = headers.authorization || headers.Authorization || ""; + return authHeader === "Bearer token-a" ? buildQuotaResponse(87) : buildQuotaResponse(20); + } + + const body = init.body ? JSON.parse(String(init.body)) : {}; + const serializedBody = JSON.stringify(body); + const isSummaryRequest = + body._omnirouteInternalRequest === "context-handoff" || + serializedBody.includes("You are a context summarizer"); + + if (isSummaryRequest) { + summaryBodies.push({ body, serializedBody }); + return buildResponsesResponse( + JSON.stringify({ + summary: "Carry over the router implementation state", + keyDecisions: ["Use global combo defaults", "Inject handoff on account switch"], + taskProgress: "Runtime and UI are wired; tests are next", + activeEntities: ["open-sse/services/combo.ts", "src/sse/handlers/chat.ts"], + }), + "gpt-5.4" + ); + } + + upstreamBodies.push({ body, serializedBody }); + return buildResponsesResponse("relay-success", "gpt-5.4"); + }; + + const firstResponse = await handleChat( + buildRequest({ + headers: { "X-Session-Id": "relay-session" }, + body: { + model: "relay-combo", + stream: false, + messages: [{ role: "user", content: "Keep implementing the new combo" }], + }, + }) + ); + const firstJson = await firstResponse.json(); + + assert.equal(firstResponse.status, 200); + assert.equal(firstJson.choices[0].message.content, "relay-success"); + + const sessionId = "ext:relay-session"; + const savedHandoff = await waitFor(() => handoffDb.getHandoff(sessionId, "relay-combo"), 2000); + assert.ok(savedHandoff); + assert.equal(savedHandoff.fromAccount, primary.id); + assert.equal(summaryBodies.length, 1); + assert.match(summaryBodies[0].serializedBody, /You are a context summarizer/); + + await providersDb.updateProviderConnection(primary.id, { + rateLimitedUntil: new Date(Date.now() + 60_000).toISOString(), + }); + + const secondResponse = await handleChat( + buildRequest({ + headers: { "X-Session-Id": "relay-session" }, + body: { + model: "relay-combo", + stream: false, + messages: [{ role: "user", content: "Continue from where you left off" }], + }, + }) + ); + const secondJson = await secondResponse.json(); + + assert.equal(secondResponse.status, 200); + assert.equal(secondJson.choices[0].message.content, "relay-success"); + assert.equal(upstreamBodies.length >= 2, true); + assert.match(upstreamBodies[1].serializedBody, //); + assert.match(upstreamBodies[1].serializedBody, /Carry over the router implementation state/); + assert.equal(handoffDb.getHandoff(sessionId, "relay-combo"), null); + await new Promise((resolve) => setTimeout(resolve, 50)); +}); diff --git a/tests/unit/combo-config.test.mjs b/tests/unit/combo-config.test.mjs index f69ed9be1f..36d2a7ee77 100644 --- a/tests/unit/combo-config.test.mjs +++ b/tests/unit/combo-config.test.mjs @@ -3,6 +3,7 @@ import assert from "node:assert/strict"; const { resolveComboConfig, getDefaultComboConfig } = await import("../../open-sse/services/comboConfig.ts"); +const { createComboSchema } = await import("../../src/shared/validation/schemas.ts"); test("getDefaultComboConfig returns a fresh copy of the defaults", () => { const first = getDefaultComboConfig(); @@ -12,6 +13,9 @@ test("getDefaultComboConfig returns a fresh copy of the defaults", () => { assert.equal(first.strategy, "priority"); assert.equal(first.maxRetries, 1); assert.equal(first.timeoutMs, 600000); + assert.equal(first.handoffThreshold, 0.85); + assert.equal(first.maxMessagesForSummary, 30); + assert.deepEqual(first.handoffProviders, ["codex"]); first.strategy = "weighted"; assert.equal(second.strategy, "priority"); @@ -77,6 +81,23 @@ test("resolveComboConfig ignores null and undefined overrides", () => { assert.equal(result.strategy, "priority"); }); +test("resolveComboConfig preserves explicit empty handoffProviders overrides", () => { + const result = resolveComboConfig( + { + config: { + handoffProviders: [], + }, + }, + { + comboDefaults: { + handoffProviders: ["codex"], + }, + } + ); + + assert.deepEqual(result.handoffProviders, []); +}); + test("resolveComboConfig skips provider overrides when provider is absent", () => { const result = resolveComboConfig( { config: {} }, @@ -95,3 +116,20 @@ test("resolveComboConfig tolerates invalid or missing inputs and falls back to d assert.deepEqual(resolveComboConfig(null, null, "openai"), getDefaultComboConfig()); assert.deepEqual(resolveComboConfig({}, { comboDefaults: null }, null), getDefaultComboConfig()); }); + +test("createComboSchema accepts context-relay strategy with handoff config", () => { + const parsed = createComboSchema.parse({ + name: "codex-relay", + models: ["codex/gpt-5.4"], + strategy: "context-relay", + config: { + handoffThreshold: 0.85, + maxMessagesForSummary: 24, + handoffModel: "", + }, + }); + + assert.equal(parsed.strategy, "context-relay"); + assert.equal(parsed.config.handoffThreshold, 0.85); + assert.equal(parsed.config.maxMessagesForSummary, 24); +}); diff --git a/tests/unit/combo-context-relay.test.mjs b/tests/unit/combo-context-relay.test.mjs new file mode 100644 index 0000000000..a15968fd05 --- /dev/null +++ b/tests/unit/combo-context-relay.test.mjs @@ -0,0 +1,373 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-combo-context-relay-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const { handleComboChat } = await import("../../open-sse/services/combo.ts"); +const core = await import("../../src/lib/db/core.ts"); +const handoffDb = await import("../../src/lib/db/contextHandoffs.ts"); +const { registerCodexConnection } = await import("../../open-sse/services/codexQuotaFetcher.ts"); +const { clearSessions, touchSession } = await import("../../open-sse/services/sessionManager.ts"); +const { resetAllComboMetrics } = await import("../../open-sse/services/comboMetrics.ts"); +const { resetAllCircuitBreakers, getCircuitBreaker } = + await import("../../src/shared/utils/circuitBreaker.ts"); +const { resetAll: resetAllSemaphores } = + await import("../../open-sse/services/rateLimitSemaphore.ts"); +const { _resetAllDecks } = await import("../../src/shared/utils/shuffleDeck.ts"); + +const originalFetch = globalThis.fetch; + +function createLog() { + const entries = []; + return { + info: (tag, msg) => entries.push({ level: "info", tag, msg }), + warn: (tag, msg) => entries.push({ level: "warn", tag, msg }), + error: (tag, msg) => entries.push({ level: "error", tag, msg }), + entries, + }; +} + +function okResponse(body = { choices: [{ message: { content: "ok" } }] }) { + return new Response(JSON.stringify(body), { + status: 200, + headers: { "content-type": "application/json" }, + }); +} + +function buildQuotaResponse(usedPercent, resetAfterSeconds = 3600) { + return new Response( + JSON.stringify({ + rate_limit: { + primary_window: { + used_percent: usedPercent, + reset_after_seconds: resetAfterSeconds, + }, + secondary_window: { + used_percent: 0, + reset_after_seconds: 86400, + }, + }, + }), + { + status: 200, + headers: { "content-type": "application/json" }, + } + ); +} + +async function resetStorage() { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +} + +async function waitFor(fn, timeoutMs = 1500) { + const startedAt = Date.now(); + while (Date.now() - startedAt < timeoutMs) { + const value = await fn(); + if (value) return value; + await new Promise((resolve) => setTimeout(resolve, 25)); + } + return null; +} + +test.beforeEach(async () => { + resetAllComboMetrics(); + resetAllCircuitBreakers(); + resetAllSemaphores(); + _resetAllDecks(); + clearSessions(); + globalThis.fetch = originalFetch; + await resetStorage(); +}); + +test.afterEach(async () => { + globalThis.fetch = originalFetch; + clearSessions(); + await new Promise((resolve) => setTimeout(resolve, 50)); +}); + +test.after(async () => { + resetAllComboMetrics(); + resetAllCircuitBreakers(); + resetAllSemaphores(); + _resetAllDecks(); + clearSessions(); + globalThis.fetch = originalFetch; + await resetStorage(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); +}); + +test("handleComboChat context-relay routes to the first available model", async () => { + const calls = []; + + const result = await handleComboChat({ + body: { + messages: [{ role: "user", content: "Hello" }], + }, + combo: { + name: "relay-first", + strategy: "context-relay", + models: ["openai/gpt-4o-mini", "claude/sonnet"], + config: { maxRetries: 0 }, + }, + handleSingleModel: async (_body, modelStr) => { + calls.push(modelStr); + return okResponse(); + }, + isModelAvailable: async () => true, + log: createLog(), + settings: null, + allCombos: null, + }); + + assert.equal(result.ok, true); + assert.deepEqual(calls, ["openai/gpt-4o-mini"]); +}); + +test("handleComboChat context-relay skips unavailable models and falls through to the next one", async () => { + const calls = []; + + const result = await handleComboChat({ + body: { + messages: [{ role: "user", content: "Fallback" }], + }, + combo: { + name: "relay-skip-unavailable", + strategy: "context-relay", + models: ["codex/gpt-5.4", "openai/gpt-4o-mini"], + config: { maxRetries: 0 }, + }, + handleSingleModel: async (_body, modelStr) => { + calls.push(modelStr); + return okResponse(); + }, + isModelAvailable: async (modelStr) => modelStr !== "codex/gpt-5.4", + log: createLog(), + settings: null, + allCombos: null, + relayOptions: { + sessionId: "sess-skip", + config: { handoffProviders: ["codex"] }, + }, + }); + + assert.equal(result.ok, true); + assert.deepEqual(calls, ["openai/gpt-4o-mini"]); +}); + +test("handleComboChat context-relay skips models with an open circuit breaker", async () => { + const breaker = getCircuitBreaker("combo:codex/gpt-5.4", { + failureThreshold: 1, + resetTimeout: 60000, + }); + breaker._onFailure(); + + const log = createLog(); + const calls = []; + + const result = await handleComboChat({ + body: { + messages: [{ role: "user", content: "Breaker" }], + }, + combo: { + name: "relay-breaker", + strategy: "context-relay", + models: ["codex/gpt-5.4", "openai/gpt-4o-mini"], + config: { maxRetries: 0 }, + }, + handleSingleModel: async (_body, modelStr) => { + calls.push(modelStr); + return okResponse(); + }, + isModelAvailable: async () => true, + log, + settings: null, + allCombos: null, + }); + + assert.equal(result.ok, true); + assert.deepEqual(calls, ["openai/gpt-4o-mini"]); + assert.ok(log.entries.some((entry) => entry.msg.includes("circuit breaker OPEN"))); +}); + +test("handleComboChat context-relay persists a handoff when codex quota reaches the warning threshold", async () => { + const sessionId = "sess-generate"; + const connectionId = "conn-generate"; + touchSession(sessionId, connectionId); + registerCodexConnection(connectionId, { + accessToken: "token-generate", + workspaceId: "ws-generate", + }); + + let usageCalls = 0; + let summaryCalls = 0; + + globalThis.fetch = async (url) => { + if (String(url).includes("/backend-api/wham/usage")) { + usageCalls += 1; + return buildQuotaResponse(87); + } + throw new Error(`Unexpected fetch: ${String(url)}`); + }; + + const result = await handleComboChat({ + body: { + messages: [{ role: "user", content: "Keep context alive" }], + }, + combo: { + name: "relay-generate", + strategy: "context-relay", + models: ["codex/gpt-5.4"], + config: { maxRetries: 0, handoffThreshold: 0.85, handoffProviders: ["codex"] }, + }, + handleSingleModel: async (body) => { + if (body._omnirouteInternalRequest === "context-handoff") { + summaryCalls += 1; + return okResponse({ + choices: [ + { + message: { + content: JSON.stringify({ + summary: "Generated from combo-level test", + keyDecisions: ["generate at 85%"], + taskProgress: "ready", + activeEntities: ["combo.ts"], + }), + }, + }, + ], + }); + } + + return okResponse(); + }, + isModelAvailable: async () => true, + log: createLog(), + settings: null, + allCombos: null, + relayOptions: { + sessionId, + config: { + handoffThreshold: 0.85, + handoffProviders: ["codex"], + }, + }, + }); + + const saved = await waitFor(() => handoffDb.getHandoff(sessionId, "relay-generate")); + + assert.equal(result.ok, true); + assert.equal(usageCalls, 1); + assert.equal(summaryCalls, 1); + assert.ok(saved); + assert.equal(saved.summary, "Generated from combo-level test"); + assert.equal(saved.fromAccount, connectionId); +}); + +test("handleComboChat context-relay respects handoffProviders and skips generation when codex is disabled", async () => { + const sessionId = "sess-disabled-provider"; + const connectionId = "conn-disabled-provider"; + touchSession(sessionId, connectionId); + registerCodexConnection(connectionId, { + accessToken: "token-disabled-provider", + workspaceId: "ws-disabled-provider", + }); + + let usageCalls = 0; + let summaryCalls = 0; + + globalThis.fetch = async (url) => { + if (String(url).includes("/backend-api/wham/usage")) { + usageCalls += 1; + return buildQuotaResponse(90); + } + throw new Error(`Unexpected fetch: ${String(url)}`); + }; + + const result = await handleComboChat({ + body: { + messages: [{ role: "user", content: "Do not generate" }], + }, + combo: { + name: "relay-disabled-provider", + strategy: "context-relay", + models: ["codex/gpt-5.4"], + config: { maxRetries: 0, handoffProviders: ["openai"] }, + }, + handleSingleModel: async (body) => { + if (body._omnirouteInternalRequest === "context-handoff") { + summaryCalls += 1; + } + return okResponse(); + }, + isModelAvailable: async () => true, + log: createLog(), + settings: null, + allCombos: null, + relayOptions: { + sessionId, + config: { + handoffProviders: ["openai"], + }, + }, + }); + + await new Promise((resolve) => setTimeout(resolve, 50)); + assert.equal(result.ok, true); + assert.equal(usageCalls, 0); + assert.equal(summaryCalls, 0); + assert.equal(handoffDb.getHandoff(sessionId, "relay-disabled-provider"), null); +}); + +test("handleComboChat context-relay treats explicit empty handoffProviders as disabled", async () => { + const sessionId = "sess-empty-providers"; + const connectionId = "conn-empty-providers"; + touchSession(sessionId, connectionId); + registerCodexConnection(connectionId, { + accessToken: "token-empty-providers", + workspaceId: "ws-empty-providers", + }); + + let usageCalls = 0; + + globalThis.fetch = async (url) => { + if (String(url).includes("/backend-api/wham/usage")) { + usageCalls += 1; + return buildQuotaResponse(91); + } + throw new Error(`Unexpected fetch: ${String(url)}`); + }; + + const result = await handleComboChat({ + body: { + messages: [{ role: "user", content: "Disabled by empty list" }], + }, + combo: { + name: "relay-empty-providers", + strategy: "context-relay", + models: ["codex/gpt-5.4"], + config: { maxRetries: 0, handoffProviders: [] }, + }, + handleSingleModel: async () => okResponse(), + isModelAvailable: async () => true, + log: createLog(), + settings: null, + allCombos: null, + relayOptions: { + sessionId, + config: { + handoffProviders: [], + }, + }, + }); + + await new Promise((resolve) => setTimeout(resolve, 50)); + assert.equal(result.ok, true); + assert.equal(usageCalls, 0); + assert.equal(handoffDb.getHandoff(sessionId, "relay-empty-providers"), null); +}); diff --git a/tests/unit/context-handoff.test.mjs b/tests/unit/context-handoff.test.mjs new file mode 100644 index 0000000000..34212219e9 --- /dev/null +++ b/tests/unit/context-handoff.test.mjs @@ -0,0 +1,336 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-context-handoff-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const handoffDb = await import("../../src/lib/db/contextHandoffs.ts"); +const contextHandoff = await import("../../open-sse/services/contextHandoff.ts"); + +async function resetStorage() { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +} + +async function waitFor(fn, timeoutMs = 1500) { + const startedAt = Date.now(); + while (Date.now() - startedAt < timeoutMs) { + const value = await fn(); + if (value) return value; + await new Promise((resolve) => setTimeout(resolve, 25)); + } + return null; +} + +test.beforeEach(async () => { + await resetStorage(); +}); + +test.after(async () => { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); +}); + +test("buildHandoffSystemMessage and injectHandoffIntoBody preserve existing history", () => { + const payload = { + sessionId: "sess-1", + comboName: "relay-combo", + fromAccount: "conn-a", + summary: "Working through combo relay integration", + keyDecisions: ["use 85% pre-handoff threshold", "delete handoff after successful replay"], + taskProgress: "Need to finish tests", + activeEntities: ["combo.ts", "chat.ts"], + messageCount: 42, + model: "codex/gpt-5.4", + warningThresholdPct: 0.85, + generatedAt: "2026-04-08T12:00:00.000Z", + expiresAt: "2026-04-08T17:00:00.000Z", + }; + const body = { + messages: [ + { role: "system", content: "Original system message" }, + { role: "user", content: "Continue" }, + ], + }; + + const systemMessage = contextHandoff.buildHandoffSystemMessage(payload); + const injected = contextHandoff.injectHandoffIntoBody(body, payload); + + assert.match(systemMessage, //); + assert.match(systemMessage, /combo\.ts/); + assert.equal(injected.messages[0].role, "system"); + assert.match(String(injected.messages[0].content), //); + assert.equal(injected.messages[1].content, "Original system message"); + assert.equal(body.messages.length, 2); +}); + +test("parseHandoffJSON accepts fenced JSON and normalizes fields", () => { + const parsed = contextHandoff.parseHandoffJSON(`\`\`\`json +{"summary":" Ready to continue ","keyDecisions":["A","B"],"taskProgress":"Pending tests","activeEntities":["file.ts","combo.ts"]} +\`\`\``); + + assert.equal(parsed.summary, "Ready to continue"); + assert.deepEqual(parsed.keyDecisions, ["A", "B"]); + assert.equal(parsed.taskProgress, "Pending tests"); + assert.deepEqual(parsed.activeEntities, ["file.ts", "combo.ts"]); +}); + +test("parseHandoffJSON returns null for invalid payloads without throwing", () => { + assert.equal(contextHandoff.parseHandoffJSON("not-json"), null); +}); + +test("resolveContextRelayConfig preserves explicit empty handoffProviders", () => { + const resolved = contextHandoff.resolveContextRelayConfig({ + handoffProviders: [], + handoffThreshold: 0.9, + maxMessagesForSummary: 15, + }); + + assert.deepEqual(resolved.handoffProviders, []); + assert.equal(resolved.handoffThreshold, 0.9); + assert.equal(resolved.maxMessagesForSummary, 15); +}); + +test("maybeGenerateHandoff skips below the warning threshold", async () => { + let called = false; + + contextHandoff.maybeGenerateHandoff({ + sessionId: "sess-low", + comboName: "relay-combo", + connectionId: "conn-low", + percentUsed: 0.7, + messages: [{ role: "user", content: "hello" }], + model: "codex/gpt-5.4", + expiresAt: null, + handleSingleModel: async () => { + called = true; + return new Response("{}", { status: 200 }); + }, + }); + + await new Promise((resolve) => setImmediate(resolve)); + assert.equal(called, false); + assert.equal(handoffDb.getHandoff("sess-low", "relay-combo"), null); +}); + +test("maybeGenerateHandoff persists a structured handoff once the threshold is reached", async () => { + const calls = []; + + contextHandoff.maybeGenerateHandoff({ + sessionId: "sess-save", + comboName: "relay-combo", + connectionId: "conn-save", + percentUsed: 0.88, + messages: [ + { role: "user", content: "Please continue wiring the combo" }, + { role: "assistant", content: "Working on it" }, + ], + model: "codex/gpt-5.4", + expiresAt: "2026-04-08T17:00:00.000Z", + handleSingleModel: async (body, modelStr) => { + calls.push({ body, modelStr }); + return new Response( + JSON.stringify({ + choices: [ + { + message: { + content: JSON.stringify({ + summary: "Relay summary generated", + keyDecisions: ["Use context-relay"], + taskProgress: "Integration in progress", + activeEntities: ["combo.ts", "contextHandoff.ts"], + }), + }, + }, + ], + }), + { + status: 200, + headers: { "content-type": "application/json" }, + } + ); + }, + }); + + const saved = await waitFor(() => handoffDb.getHandoff("sess-save", "relay-combo")); + assert.ok(saved); + assert.equal(saved.fromAccount, "conn-save"); + assert.equal(saved.summary, "Relay summary generated"); + assert.deepEqual(saved.keyDecisions, ["Use context-relay"]); + assert.equal(calls.length, 1); + assert.equal(calls[0].modelStr, "codex/gpt-5.4"); + assert.equal(calls[0].body._omnirouteSkipContextRelay, true); + assert.equal(calls[0].body._omnirouteInternalRequest, "context-handoff"); +}); + +test("maybeGenerateHandoff deduplicates concurrent in-flight generations for the same session", async () => { + const calls = []; + let releaseGeneration; + const gate = new Promise((resolve) => { + releaseGeneration = resolve; + }); + + const options = { + sessionId: "sess-dedupe", + comboName: "relay-combo", + connectionId: "conn-dedupe", + percentUsed: 0.89, + messages: [{ role: "user", content: "Generate once" }], + model: "codex/gpt-5.4", + expiresAt: "2099-01-01T00:00:00.000Z", + handleSingleModel: async () => { + calls.push("summary"); + await gate; + return new Response( + JSON.stringify({ + choices: [ + { + message: { + content: JSON.stringify({ + summary: "Only one handoff should be generated", + keyDecisions: ["dedupe in flight"], + taskProgress: "done", + activeEntities: ["contextHandoff.ts"], + }), + }, + }, + ], + }), + { + status: 200, + headers: { "content-type": "application/json" }, + } + ); + }, + }; + + contextHandoff.maybeGenerateHandoff(options); + contextHandoff.maybeGenerateHandoff(options); + + await new Promise((resolve) => setTimeout(resolve, 25)); + assert.equal(calls.length, 1); + + releaseGeneration(); + const saved = await waitFor(() => handoffDb.getHandoff("sess-dedupe", "relay-combo")); + assert.ok(saved); + assert.equal(saved.summary, "Only one handoff should be generated"); + assert.equal(calls.length, 1); +}); + +test("maybeGenerateHandoff allows a new attempt after a failed in-flight generation", async () => { + let calls = 0; + + const options = { + sessionId: "sess-retry", + comboName: "relay-combo", + connectionId: "conn-retry", + percentUsed: 0.9, + messages: [{ role: "user", content: "Retry after failure" }], + model: "codex/gpt-5.4", + expiresAt: "2099-01-01T00:00:00.000Z", + handleSingleModel: async () => { + calls += 1; + if (calls === 1) { + return new Response("temporary failure", { status: 500 }); + } + + return new Response( + JSON.stringify({ + choices: [ + { + message: { + content: JSON.stringify({ + summary: "Retry succeeded", + keyDecisions: ["lock cleared after failure"], + taskProgress: "completed", + activeEntities: ["contextHandoff.ts"], + }), + }, + }, + ], + }), + { + status: 200, + headers: { "content-type": "application/json" }, + } + ); + }, + }; + + contextHandoff.maybeGenerateHandoff(options); + await new Promise((resolve) => setTimeout(resolve, 40)); + assert.equal(handoffDb.getHandoff("sess-retry", "relay-combo"), null); + + contextHandoff.maybeGenerateHandoff(options); + const saved = await waitFor(() => handoffDb.getHandoff("sess-retry", "relay-combo")); + assert.ok(saved); + assert.equal(saved.summary, "Retry succeeded"); + assert.equal(calls, 2); +}); + +test("maybeGenerateHandoff respects explicit empty handoffProviders and skips generation", async () => { + let called = false; + + contextHandoff.maybeGenerateHandoff({ + sessionId: "sess-disabled", + comboName: "relay-combo", + connectionId: "conn-disabled", + percentUsed: 0.92, + messages: [{ role: "user", content: "Do not generate" }], + model: "codex/gpt-5.4", + expiresAt: null, + config: { handoffProviders: [] }, + handleSingleModel: async () => { + called = true; + return new Response("{}", { status: 200 }); + }, + }); + + await new Promise((resolve) => setImmediate(resolve)); + assert.equal(called, false); + assert.equal(handoffDb.getHandoff("sess-disabled", "relay-combo"), null); +}); + +test("context handoff DB module upserts and deletes active handoffs", () => { + handoffDb.upsertHandoff({ + sessionId: "sess-db", + comboName: "relay-combo", + fromAccount: "conn-a", + summary: "First summary", + keyDecisions: ["A"], + taskProgress: "step one", + activeEntities: ["a.ts"], + messageCount: 3, + model: "codex/gpt-5.4", + warningThresholdPct: 0.85, + generatedAt: "2026-04-08T10:00:00.000Z", + expiresAt: "2099-01-01T00:00:00.000Z", + }); + handoffDb.upsertHandoff({ + sessionId: "sess-db", + comboName: "relay-combo", + fromAccount: "conn-b", + summary: "Updated summary", + keyDecisions: ["B"], + taskProgress: "step two", + activeEntities: ["b.ts"], + messageCount: 4, + model: "codex/gpt-5.4", + warningThresholdPct: 0.86, + generatedAt: "2026-04-08T11:00:00.000Z", + expiresAt: "2099-01-01T00:00:00.000Z", + }); + + const saved = handoffDb.getHandoff("sess-db", "relay-combo"); + assert.equal(saved.fromAccount, "conn-b"); + assert.equal(saved.summary, "Updated summary"); + assert.equal(handoffDb.hasActiveHandoff("sess-db", "relay-combo"), true); + + handoffDb.deleteHandoff("sess-db", "relay-combo"); + assert.equal(handoffDb.getHandoff("sess-db", "relay-combo"), null); +});